{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/10","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":10,"pages_in_order":135,"rows_per_page":100,"rows":[901,1000],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/9","next":"/task/reinforcement-learning-2/papers/11","papers":[{"url":"/paper/hypercontroller-a-hyperparameter-controller","slug":"hypercontroller-a-hyperparameter-controller","title":"HyperController: A Hyperparameter Controller for Fast and Stable Training of Reinforcement Learning Neural Networks","date":"2025-04-27","arxiv_id":"2504.19382","repositories_listed":1,"syntology":null},{"url":"/paper/skywork-r1v2-multimodal-hybrid-reinforcement","slug":"skywork-r1v2-multimodal-hybrid-reinforcement","title":"Skywork R1V2: Multimodal Hybrid Reinforcement Learning for Reasoning","date":"2025-04-23","arxiv_id":"2504.16656","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/skywork-r1v2-multimodal-hybrid-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2504.16656","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16656"}},"official":{"repos":["SkyworkAI/Skywork-R1V"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/compile-scene-graphs-with-reinforcement","slug":"compile-scene-graphs-with-reinforcement","title":"Compile Scene Graphs with Reinforcement Learning","date":"2025-04-18","arxiv_id":"2504.13617","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/compile-scene-graphs-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2504.13617","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13617"}},"official":{"repos":["gpt4vision/r1-sgg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-from-human-feedback-4","slug":"reinforcement-learning-from-human-feedback-4","title":"Reinforcement Learning from Human Feedback","date":"2025-04-16","arxiv_id":"2504.12501","repositories_listed":1,"syntology":null},{"url":"/paper/a-pytorch-compatible-spike-encoding-framework","slug":"a-pytorch-compatible-spike-encoding-framework","title":"A PyTorch-Compatible Spike Encoding Framework for Energy-Efficient Neuromorphic Applications","date":"2025-04-15","arxiv_id":"2504.11026","repositories_listed":1,"syntology":null},{"url":"/paper/retool-reinforcement-learning-for-strategic","slug":"retool-reinforcement-learning-for-strategic","title":"ReTool: Reinforcement Learning for Strategic Tool Use in LLMs","date":"2025-04-15","arxiv_id":"2504.11536","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-sensor-steering-strategy-using-deep","slug":"adaptive-sensor-steering-strategy-using-deep","title":"Adaptive Sensor Steering Strategy Using Deep Reinforcement Learning for Dynamic Data Acquisition in Digital Twins","date":"2025-04-14","arxiv_id":"2504.10248","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reasoning-translation-via-reinforcement","slug":"deep-reasoning-translation-via-reinforcement","title":"Deep Reasoning Translation via Reinforcement Learning","date":"2025-04-14","arxiv_id":"2504.10187","repositories_listed":1,"syntology":null},{"url":"/paper/pay-attention-to-what-and-where-interpretable","slug":"pay-attention-to-what-and-where-interpretable","title":"Pay Attention to What and Where? Interpretable Feature Extractor in Vision-based Deep Reinforcement Learning","date":"2025-04-14","arxiv_id":"2504.10071","repositories_listed":1,"syntology":null},{"url":"/paper/tinyllava-video-r1-towards-smaller-lmms-for","slug":"tinyllava-video-r1-towards-smaller-lmms-for","title":"TinyLLaVA-Video-R1: Towards Smaller LMMs for Video Reasoning","date":"2025-04-13","arxiv_id":"2504.09641","repositories_listed":1,"syntology":null},{"url":"/paper/interq-a-dqn-framework-for-optimal","slug":"interq-a-dqn-framework-for-optimal","title":"InterQ: A DQN Framework for Optimal Intermittent Control","date":"2025-04-12","arxiv_id":"2504.09035","repositories_listed":1,"syntology":null},{"url":"/paper/perception-r1-pioneering-perception-policy","slug":"perception-r1-pioneering-perception-policy","title":"Perception-R1: Pioneering Perception Policy with Reinforcement Learning","date":"2025-04-10","arxiv_id":"2504.07954","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/perception-r1-pioneering-perception-policy#ran","syntology_url":"https://syntology.ai/paper/2504.07954","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07954"}},"official":{"repos":["linkangheng/pr1"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vlm-r1-a-stable-and-generalizable-r1-style","slug":"vlm-r1-a-stable-and-generalizable-r1-style","title":"VLM-R1: A Stable and Generalizable R1-style Large Vision-Language Model","date":"2025-04-10","arxiv_id":"2504.07615","repositories_listed":1,"syntology":null},{"url":"/paper/free-random-projection-for-in-context","slug":"free-random-projection-for-in-context","title":"Free Random Projection for In-Context Reinforcement Learning","date":"2025-04-09","arxiv_id":"2504.06983","repositories_listed":1,"syntology":null},{"url":"/paper/neural-motion-simulator-pushing-the-limit-of","slug":"neural-motion-simulator-pushing-the-limit-of","title":"Neural Motion Simulator: Pushing the Limit of World Models in Reinforcement Learning","date":"2025-04-09","arxiv_id":"2504.07095","repositories_listed":1,"syntology":null},{"url":"/paper/leanabell-prover-posttraining-scaling-in","slug":"leanabell-prover-posttraining-scaling-in","title":"Leanabell-Prover: Posttraining Scaling in Formal Reasoning","date":"2025-04-08","arxiv_id":"2504.06122","repositories_listed":1,"syntology":null},{"url":"/paper/robo-taxi-fleet-coordination-at-scale-via","slug":"robo-taxi-fleet-coordination-at-scale-via","title":"Robo-taxi Fleet Coordination at Scale via Reinforcement Learning","date":"2025-04-08","arxiv_id":"2504.06125","repositories_listed":1,"syntology":null},{"url":"/paper/concise-reasoning-via-reinforcement-learning","slug":"concise-reasoning-via-reinforcement-learning","title":"Concise Reasoning via Reinforcement Learning","date":"2025-04-07","arxiv_id":"2504.05185","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":10,"n_pointer_only":13,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/concise-reasoning-via-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2504.05185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.05185"}},"official":{"repos":["ai-wand/concise-reasoning"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/deep-reinforcement-learning-algorithms-for-1","slug":"deep-reinforcement-learning-algorithms-for-1","title":"Deep Reinforcement Learning Algorithms for Option Hedging","date":"2025-04-07","arxiv_id":"2504.05521","repositories_listed":1,"syntology":null},{"url":"/paper/large-scale-mixed-traffic-and-intersection","slug":"large-scale-mixed-traffic-and-intersection","title":"Large-Scale Mixed-Traffic and Intersection Control using Multi-agent Reinforcement Learning","date":"2025-04-07","arxiv_id":"2504.04691","repositories_listed":1,"syntology":null},{"url":"/paper/playing-non-embedded-card-based-games-with","slug":"playing-non-embedded-card-based-games-with","title":"Playing Non-Embedded Card-Based Games with Reinforcement Learning","date":"2025-04-07","arxiv_id":"2504.04783","repositories_listed":1,"syntology":null},{"url":"/paper/ai2stow-end-to-end-deep-reinforcement","slug":"ai2stow-end-to-end-deep-reinforcement","title":"AI2STOW: End-to-End Deep Reinforcement Learning to Construct Master Stowage Plans under Demand Uncertainty","date":"2025-04-06","arxiv_id":"2504.04469","repositories_listed":1,"syntology":null},{"url":"/paper/solving-sokoban-using-hierarchical-1","slug":"solving-sokoban-using-hierarchical-1","title":"Solving Sokoban using Hierarchical Reinforcement Learning with Landmarks","date":"2025-04-06","arxiv_id":"2504.04366","repositories_listed":1,"syntology":null},{"url":"/paper/distillation-and-refinement-of-reasoning-in","slug":"distillation-and-refinement-of-reasoning-in","title":"Distillation and Refinement of Reasoning in Small Language Models for Document Re-ranking","date":"2025-04-04","arxiv_id":"2504.03947","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-via-object","slug":"deep-reinforcement-learning-via-object","title":"Deep Reinforcement Learning via Object-Centric Attention","date":"2025-04-03","arxiv_id":"2504.03024","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-policy-gradient-reinforcement","slug":"hierarchical-policy-gradient-reinforcement","title":"Hierarchical Policy-Gradient Reinforcement Learning for Multi-Agent Shepherding Control of Non-Cohesive Targets","date":"2025-04-03","arxiv_id":"2504.02479","repositories_listed":1,"syntology":null},{"url":"/paper/low-rank-factorizations-are-indirect","slug":"low-rank-factorizations-are-indirect","title":"Low Rank Factorizations are Indirect Encodings for Deep Neuroevolution","date":"2025-04-03","arxiv_id":"2504.03037","repositories_listed":1,"syntology":null},{"url":"/paper/probabilistic-pontryagin-s-maximum-principle","slug":"probabilistic-pontryagin-s-maximum-principle","title":"Probabilistic Pontryagin's Maximum Principle for Continuous-Time Model-Based Reinforcement Learning","date":"2025-04-03","arxiv_id":"2504.02543","repositories_listed":1,"syntology":null},{"url":"/paper/probabilistically-safe-and-efficient-model","slug":"probabilistically-safe-and-efficient-model","title":"Probabilistically safe and efficient model-based Reinforcement Learning","date":"2025-04-01","arxiv_id":"2504.00626","repositories_listed":1,"syntology":null},{"url":"/paper/handling-delay-in-real-time-reinforcement","slug":"handling-delay-in-real-time-reinforcement","title":"Handling Delay in Real-Time Reinforcement Learning","date":"2025-03-30","arxiv_id":"2503.23478","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":2,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":7,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/handling-delay-in-real-time-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2503.23478","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23478"}},"official":{"repos":["avecplezir/realtime-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/torl-scaling-tool-integrated-rl","slug":"torl-scaling-tool-integrated-rl","title":"ToRL: Scaling Tool-Integrated RL","date":"2025-03-30","arxiv_id":"2503.23383","repositories_listed":1,"syntology":null},{"url":"/paper/q-insight-understanding-image-quality-via","slug":"q-insight-understanding-image-quality-via","title":"Q-Insight: Understanding Image Quality via Visual Reinforcement Learning","date":"2025-03-28","arxiv_id":"2503.22679","repositories_listed":1,"syntology":null},{"url":"/paper/reward-design-for-reinforcement-learning","slug":"reward-design-for-reinforcement-learning","title":"Reward Design for Reinforcement Learning Agents","date":"2025-03-27","arxiv_id":"2503.21949","repositories_listed":1,"syntology":null},{"url":"/paper/neorl-2-near-real-world-benchmarks-for","slug":"neorl-2-near-real-world-benchmarks-for","title":"NeoRL-2: Near Real-World Benchmarks for Offline Reinforcement Learning with Extended Realistic Scenarios","date":"2025-03-25","arxiv_id":"2503.19267","repositories_listed":1,"syntology":null},{"url":"/paper/research-learning-to-reason-with-search-for","slug":"research-learning-to-reason-with-search-for","title":"ReSearch: Learning to Reason with Search for LLMs via Reinforcement Learning","date":"2025-03-25","arxiv_id":"2503.19470","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/research-learning-to-reason-with-search-for#ran","syntology_url":"https://syntology.ai/paper/2503.19470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.19470"}},"official":null}},{"url":"/paper/continual-reinforcement-learning-for-hvac","slug":"continual-reinforcement-learning-for-hvac","title":"Continual Reinforcement Learning for HVAC Systems Control: Integrating Hypernetworks and Transfer Learning","date":"2025-03-24","arxiv_id":"2503.19212","repositories_listed":1,"syntology":null},{"url":"/paper/curriculum-rl-meets-monte-carlo-planning","slug":"curriculum-rl-meets-monte-carlo-planning","title":"Curriculum RL meets Monte Carlo Planning: Optimization of a Real World Container Management Problem","date":"2025-03-21","arxiv_id":"2503.17194","repositories_listed":1,"syntology":null},{"url":"/paper/fastcurl-curriculum-reinforcement-learning","slug":"fastcurl-curriculum-reinforcement-learning","title":"FastCuRL: Curriculum Reinforcement Learning with Progressive Context Extension for Efficient Training R1-like Reasoning Models","date":"2025-03-21","arxiv_id":"2503.17287","repositories_listed":1,"syntology":null},{"url":"/paper/neural-guided-equation-discovery","slug":"neural-guided-equation-discovery","title":"Neural-Guided Equation Discovery","date":"2025-03-21","arxiv_id":"2503.16953","repositories_listed":1,"syntology":null},{"url":"/paper/cls-rl-image-classification-with-rule-based","slug":"cls-rl-image-classification-with-rule-based","title":"Think or Not Think: A Study of Explicit Thinking in Rule-Based Visual Reinforcement Fine-Tuning","date":"2025-03-20","arxiv_id":"2503.16188","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cls-rl-image-classification-with-rule-based#ran","syntology_url":"https://syntology.ai/paper/2503.16188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16188"}},"official":{"repos":["minglllli/CLS-RL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-based-heuristics-to","slug":"reinforcement-learning-based-heuristics-to","title":"Reinforcement Learning-based Heuristics to Guide Domain-Independent Dynamic Programming","date":"2025-03-20","arxiv_id":"2503.16371","repositories_listed":1,"syntology":null},{"url":"/paper/application-of-linear-regression-method-to","slug":"application-of-linear-regression-method-to","title":"Application of linear regression method to the deep reinforcement learning in continuous action cases","date":"2025-03-19","arxiv_id":"2503.14976","repositories_listed":1,"syntology":null},{"url":"/paper/had-gen-human-like-and-diverse-driving","slug":"had-gen-human-like-and-diverse-driving","title":"HAD-Gen: Human-like and Diverse Driving Behavior Modeling for Controllable Scenario Generation","date":"2025-03-19","arxiv_id":"2503.15049","repositories_listed":1,"syntology":null},{"url":"/paper/learning-with-expert-abstractions-for","slug":"learning-with-expert-abstractions-for","title":"Learning with Expert Abstractions for Efficient Multi-Task Continuous Control","date":"2025-03-19","arxiv_id":"2503.14809","repositories_listed":1,"syntology":null},{"url":"/paper/neural-lyapunov-function-approximation-with","slug":"neural-lyapunov-function-approximation-with","title":"Neural Lyapunov Function Approximation with Self-Supervised Reinforcement Learning","date":"2025-03-19","arxiv_id":"2503.15629","repositories_listed":1,"syntology":null},{"url":"/paper/counterfactual-experience-augmented-off","slug":"counterfactual-experience-augmented-off","title":"Counterfactual experience augmented off-policy reinforcement learning","date":"2025-03-18","arxiv_id":"2503.13842","repositories_listed":1,"syntology":null},{"url":"/paper/dapo-an-open-source-llm-reinforcement","slug":"dapo-an-open-source-llm-reinforcement","title":"DAPO: An Open-Source LLM Reinforcement Learning System at Scale","date":"2025-03-18","arxiv_id":"2503.14476","repositories_listed":1,"syntology":null},{"url":"/paper/socialjax-an-evaluation-suite-for-multi-agent","slug":"socialjax-an-evaluation-suite-for-multi-agent","title":"SocialJax: An Evaluation Suite for Multi-agent Reinforcement Learning in Sequential Social Dilemmas","date":"2025-03-18","arxiv_id":"2503.14576","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/socialjax-an-evaluation-suite-for-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2503.14576","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.14576"}},"official":{"repos":["cooperativex/socialjax"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ipcgrl-language-instructed-reinforcement","slug":"ipcgrl-language-instructed-reinforcement","title":"IPCGRL: Language-Instructed Reinforcement Learning for Procedural Level Generation","date":"2025-03-16","arxiv_id":"2503.12358","repositories_listed":1,"syntology":null},{"url":"/paper/terl-large-scale-multi-target-encirclement","slug":"terl-large-scale-multi-target-encirclement","title":"TERL: Large-Scale Multi-Target Encirclement Using Transformer-Enhanced Reinforcement Learning","date":"2025-03-16","arxiv_id":"2503.12395","repositories_listed":1,"syntology":null},{"url":"/paper/from-demonstrations-to-rewards-alignment","slug":"from-demonstrations-to-rewards-alignment","title":"From Demonstrations to Rewards: Alignment Without Explicit Human Preferences","date":"2025-03-15","arxiv_id":"2503.13538","repositories_listed":1,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/from-demonstrations-to-rewards-alignment#ran","syntology_url":"https://syntology.ai/paper/2503.13538","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13538"}},"official":{"repos":["Hong-Lab-UMN-ECE/IRLAlignment"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-optimal-offline-reinforcement","slug":"towards-optimal-offline-reinforcement","title":"Towards Optimal Offline Reinforcement Learning","date":"2025-03-15","arxiv_id":"2503.12283","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-reset-in-target-search-problems","slug":"learning-to-reset-in-target-search-problems","title":"Learning to reset in target search problems","date":"2025-03-14","arxiv_id":"2503.11330","repositories_listed":1,"syntology":null},{"url":"/paper/rema-learning-to-meta-think-for-llms-with","slug":"rema-learning-to-meta-think-for-llms-with","title":"ReMA: Learning to Meta-think for LLMs with Multi-Agent Reinforcement Learning","date":"2025-03-12","arxiv_id":"2503.09501","repositories_listed":1,"syntology":null},{"url":"/paper/regulatory-dna-sequence-design-with","slug":"regulatory-dna-sequence-design-with","title":"Regulatory DNA sequence Design with Reinforcement Learning","date":"2025-03-11","arxiv_id":"2503.07981","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regulatory-dna-sequence-design-with#ran","syntology_url":"https://syntology.ai/paper/2503.07981","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.07981"}},"official":{"repos":["yangzhao1230/taco"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/v-max-making-rl-practical-for-autonomous","slug":"v-max-making-rl-practical-for-autonomous","title":"V-Max: A Reinforcement Learning Framework for Autonomous Driving","date":"2025-03-11","arxiv_id":"2503.08388","repositories_listed":1,"syntology":null},{"url":"/paper/dad-distilled-reinforcement-learning-for","slug":"dad-distilled-reinforcement-learning-for","title":"DaD: Distilled Reinforcement Learning for Diverse Keypoint Detection","date":"2025-03-10","arxiv_id":"2503.07347","repositories_listed":1,"syntology":null},{"url":"/paper/automated-proof-of-polynomial-inequalities","slug":"automated-proof-of-polynomial-inequalities","title":"Automated Proof of Polynomial Inequalities via Reinforcement Learning","date":"2025-03-09","arxiv_id":"2503.06592","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/automated-proof-of-polynomial-inequalities#ran","syntology_url":"https://syntology.ai/paper/2503.06592","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.06592"}},"official":{"repos":["blliu6/appirl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamics-invariant-quadrotor-control-using","slug":"dynamics-invariant-quadrotor-control-using","title":"Dynamics-Invariant Quadrotor Control using Scale-Aware Deep Reinforcement Learning","date":"2025-03-09","arxiv_id":"2503.09622","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-policy-optimization-for-offline","slug":"adversarial-policy-optimization-for-offline","title":"Adversarial Policy Optimization for Offline Preference-based Reinforcement Learning","date":"2025-03-07","arxiv_id":"2503.05306","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adversarial-policy-optimization-for-offline#ran","syntology_url":"https://syntology.ai/paper/2503.05306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05306"}},"official":{"repos":["oh-lab/APPO"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/performance-comparisons-of-reinforcement","slug":"performance-comparisons-of-reinforcement","title":"Performance Comparisons of Reinforcement Learning Algorithms for Sequential Experimental Design","date":"2025-03-07","arxiv_id":"2503.05905","repositories_listed":1,"syntology":null},{"url":"/paper/policy-constraint-by-only-support-constraint","slug":"policy-constraint-by-only-support-constraint","title":"Policy Constraint by Only Support Constraint for Offline Reinforcement Learning","date":"2025-03-07","arxiv_id":"2503.05207","repositories_listed":1,"syntology":null},{"url":"/paper/r1-omni-explainable-omni-multimodal-emotion","slug":"r1-omni-explainable-omni-multimodal-emotion","title":"R1-Omni: Explainable Omni-Multimodal Emotion Recognition with Reinforcement Learning","date":"2025-03-07","arxiv_id":"2503.05379","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"0 ran · 3 unverified","sample_list":"/paper/r1-omni-explainable-omni-multimodal-emotion#ran","syntology_url":"https://syntology.ai/paper/2503.05379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05379"}},"official":null}},{"url":"/paper/r1-zero-s-aha-moment-in-visual-reasoning-on-a","slug":"r1-zero-s-aha-moment-in-visual-reasoning-on-a","title":"R1-Zero's \"Aha Moment\" in Visual Reasoning on a 2B Non-SFT Model","date":"2025-03-07","arxiv_id":"2503.05132","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/r1-zero-s-aha-moment-in-visual-reasoning-on-a#ran","syntology_url":"https://syntology.ai/paper/2503.05132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.05132"}},"official":{"repos":["turningpoint-ai/visualthinker-r1-zero"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/experience-replay-with-random-reshuffling","slug":"experience-replay-with-random-reshuffling","title":"Experience Replay with Random Reshuffling","date":"2025-03-04","arxiv_id":"2503.02269","repositories_listed":1,"syntology":null},{"url":"/paper/trajectory-class-aware-multi-agent","slug":"trajectory-class-aware-multi-agent","title":"Trajectory-Class-Aware Multi-Agent Reinforcement Learning","date":"2025-03-03","arxiv_id":"2503.01440","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/trajectory-class-aware-multi-agent#ran","syntology_url":"https://syntology.ai/paper/2503.01440","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.01440"}},"official":{"repos":["aailab-kaist/trama"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/on-generalization-across-environments-in","slug":"on-generalization-across-environments-in","title":"On Generalization Across Environments In Multi-Objective Reinforcement Learning","date":"2025-03-02","arxiv_id":"2503.00799","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-combinatorial-1","slug":"reinforcement-learning-with-combinatorial-1","title":"Reinforcement learning with combinatorial actions for coupled restless bandits","date":"2025-03-01","arxiv_id":"2503.01919","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-combinatorial-1#ran","syntology_url":"https://syntology.ai/paper/2503.01919","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.01919"}},"official":{"repos":["lily-x/combinatorial-rmab"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deepretrieval-powerful-query-generation-for","slug":"deepretrieval-powerful-query-generation-for","title":"DeepRetrieval: Hacking Real Search Engines and Retrievers with Large Language Models via Reinforcement Learning","date":"2025-02-28","arxiv_id":"2503.00223","repositories_listed":1,"syntology":{"n":25,"n_ran":24,"n_constructed":0,"n_ran_checked":23,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":23,"n_pointer_only":0,"phrase":"24 ran (of which 0 constructed an object rather than computing a result; 23 with no instrument failure: 0 honoured, 0 violated, 23 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deepretrieval-powerful-query-generation-for#ran","syntology_url":"https://syntology.ai/paper/2503.00223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.00223"}},"official":{"repos":["pat-jj/deepretrieval"],"state":"official (archive's flag): 24 ran","n_ran":24,"n_constructed":0,"n_ran_no_instrument_failure":23,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/highly-parallelized-reinforcement-learning","slug":"highly-parallelized-reinforcement-learning","title":"Highly Parallelized Reinforcement Learning Training with Relaxed Assignment Dependencies","date":"2025-02-27","arxiv_id":"2502.20190","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-via-inverse","slug":"offline-reinforcement-learning-via-inverse","title":"Offline Reinforcement Learning via Inverse Optimization","date":"2025-02-27","arxiv_id":"2502.20030","repositories_listed":1,"syntology":null},{"url":"/paper/playing-pokemon-red-via-deep-reinforcement","slug":"playing-pokemon-red-via-deep-reinforcement","title":"Playing Pokémon Red via Deep Reinforcement Learning","date":"2025-02-27","arxiv_id":"2502.19920","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/playing-pokemon-red-via-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2502.19920","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.19920"}},"official":{"repos":["MarcoMeter/neroRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rize-regularized-imitation-learning-via","slug":"rize-regularized-imitation-learning-via","title":"RIZE: Regularized Imitation Learning via Distributional Reinforcement Learning","date":"2025-02-27","arxiv_id":"2502.20089","repositories_listed":1,"syntology":null},{"url":"/paper/controlling-dynamics-of-stochastic-systems","slug":"controlling-dynamics-of-stochastic-systems","title":"Controlling dynamics of stochastic systems with deep reinforcement learning","date":"2025-02-25","arxiv_id":"2502.18111","repositories_listed":1,"syntology":null},{"url":"/paper/notagen-advancing-musicality-in-symbolic","slug":"notagen-advancing-musicality-in-symbolic","title":"NotaGen: Advancing Musicality in Symbolic Music Generation with Large Language Model Training Paradigms","date":"2025-02-25","arxiv_id":"2502.18008","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/notagen-advancing-musicality-in-symbolic#ran","syntology_url":"https://syntology.ai/paper/2502.18008","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.18008"}},"official":null}},{"url":"/paper/tdmpbc-self-imitative-reinforcement-learning","slug":"tdmpbc-self-imitative-reinforcement-learning","title":"TDMPBC: Self-Imitative Reinforcement Learning for Humanoid Robot Control","date":"2025-02-24","arxiv_id":"2502.17322","repositories_listed":1,"syntology":null},{"url":"/paper/pmat-optimizing-action-generation-order-in","slug":"pmat-optimizing-action-generation-order-in","title":"PMAT: Optimizing Action Generation Order in Multi-Agent Reinforcement Learning","date":"2025-02-23","arxiv_id":"2502.16496","repositories_listed":1,"syntology":null},{"url":"/paper/risk-averse-reinforcement-learning-an-optimal","slug":"risk-averse-reinforcement-learning-an-optimal","title":"Risk-Averse Reinforcement Learning: An Optimal Transport Perspective on Temporal Difference Learning","date":"2025-02-22","arxiv_id":"2502.16328","repositories_listed":1,"syntology":null},{"url":"/paper/statistical-inference-in-reinforcement","slug":"statistical-inference-in-reinforcement","title":"Statistical Inference in Reinforcement Learning: A Selective Survey","date":"2025-02-22","arxiv_id":"2502.16195","repositories_listed":1,"syntology":null},{"url":"/paper/generating-p-functional-molecules-using-stgg","slug":"generating-p-functional-molecules-using-stgg","title":"Generating $π$-Functional Molecules Using STGG+ with Active Learning","date":"2025-02-20","arxiv_id":"2502.14842","repositories_listed":1,"syntology":null},{"url":"/paper/navigating-demand-uncertainty-in-container","slug":"navigating-demand-uncertainty-in-container","title":"Navigating Demand Uncertainty in Container Shipping: Deep Reinforcement Learning for Enabling Adaptive and Feasible Master Stowage Planning","date":"2025-02-18","arxiv_id":"2502.12756","repositories_listed":1,"syntology":null},{"url":"/paper/maximum-entropy-reinforcement-learning-with-1","slug":"maximum-entropy-reinforcement-learning-with-1","title":"Maximum Entropy Reinforcement Learning with Diffusion Policy","date":"2025-02-17","arxiv_id":"2502.11612","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":6,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":13,"phrase":"10 ran (of which 6 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/maximum-entropy-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2502.11612","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.11612"}},"official":{"repos":["diffusionyes/maxentdp"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":6,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-the-paperclip-maximizer-are-rl","slug":"evaluating-the-paperclip-maximizer-are-rl","title":"Evaluating the Paperclip Maximizer: Are RL-Based Language Models More Likely to Pursue Instrumental Goals?","date":"2025-02-16","arxiv_id":"2502.12206","repositories_listed":1,"syntology":null},{"url":"/paper/reinforced-large-language-model-is-a-formal","slug":"reinforced-large-language-model-is-a-formal","title":"Reinforced Large Language Model is a formal theorem prover","date":"2025-02-13","arxiv_id":"2502.08908","repositories_listed":1,"syntology":null},{"url":"/paper/active-advantage-aligned-online-reinforcement","slug":"active-advantage-aligned-online-reinforcement","title":"Active Advantage-Aligned Online Reinforcement Learning with Offline Data","date":"2025-02-11","arxiv_id":"2502.07937","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/active-advantage-aligned-online-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2502.07937","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.07937"}},"official":{"repos":["xuefeng-cs/a3rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-constraint-violation-signals-for","slug":"leveraging-constraint-violation-signals-for","title":"Leveraging Constraint Violation Signals For Action-Constrained Reinforcement Learning","date":"2025-02-08","arxiv_id":"2502.10431","repositories_listed":1,"syntology":null},{"url":"/paper/mol-moe-training-preference-guided-routers","slug":"mol-moe-training-preference-guided-routers","title":"Mol-MoE: Training Preference-Guided Routers for Molecule Generation","date":"2025-02-08","arxiv_id":"2502.05633","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mol-moe-training-preference-guided-routers#ran","syntology_url":"https://syntology.ai/paper/2502.05633","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.05633"}},"official":{"repos":["ddidacus/mol-moe"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-convergence-and-stability-of-upside","slug":"on-the-convergence-and-stability-of-upside","title":"On the Convergence and Stability of Upside-Down Reinforcement Learning, Goal-Conditioned Supervised Learning, and Online Decision Transformers","date":"2025-02-08","arxiv_id":"2502.05672","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-graph-of-thoughts-test-time-adaptive","slug":"adaptive-graph-of-thoughts-test-time-adaptive","title":"Adaptive Graph of Thoughts: Test-Time Adaptive Reasoning Unifying Chain, Tree, and Graph Structures","date":"2025-02-07","arxiv_id":"2502.05078","repositories_listed":1,"syntology":null},{"url":"/paper/deep-meta-coordination-graphs-for-multi-agent","slug":"deep-meta-coordination-graphs-for-multi-agent","title":"Deep Meta Coordination Graphs for Multi-agent Reinforcement Learning","date":"2025-02-06","arxiv_id":"2502.04028","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-with-focal","slug":"multi-agent-reinforcement-learning-with-focal","title":"Multi-Agent Reinforcement Learning with Focal Diversity Optimization","date":"2025-02-06","arxiv_id":"2502.04492","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/multi-agent-reinforcement-learning-with-focal#ran","syntology_url":"https://syntology.ai/paper/2502.04492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.04492"}},"official":{"repos":["sftekin/rl-focal"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/interactive-symbolic-regression-through","slug":"interactive-symbolic-regression-through","title":"Interactive Symbolic Regression through Offline Reinforcement Learning: A Co-Design Framework","date":"2025-02-05","arxiv_id":"2502.02917","repositories_listed":1,"syntology":null},{"url":"/paper/wolfpack-adversarial-attack-for-robust-multi","slug":"wolfpack-adversarial-attack-for-robust-multi","title":"Wolfpack Adversarial Attack for Robust Multi-Agent Reinforcement Learning","date":"2025-02-05","arxiv_id":"2502.02844","repositories_listed":1,"syntology":{"n":7,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/wolfpack-adversarial-attack-for-robust-multi#ran","syntology_url":"https://syntology.ai/paper/2502.02844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02844"}},"official":{"repos":["sunwoolee0504/wall"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/reusing-embeddings-reproducible-reward-model","slug":"reusing-embeddings-reproducible-reward-model","title":"Reusing Embeddings: Reproducible Reward Model Research in Large Language Model Alignment without GPUs","date":"2025-02-04","arxiv_id":"2502.04357","repositories_listed":1,"syntology":null},{"url":"/paper/fedhpd-heterogeneous-federated-reinforcement","slug":"fedhpd-heterogeneous-federated-reinforcement","title":"FedHPD: Heterogeneous Federated Reinforcement Learning via Policy Distillation","date":"2025-02-02","arxiv_id":"2502.00870","repositories_listed":1,"syntology":null},{"url":"/paper/sharpie-a-modular-framework-for-reinforcement","slug":"sharpie-a-modular-framework-for-reinforcement","title":"SHARPIE: A Modular Framework for Reinforcement Learning and Human-AI Interaction Experiments","date":"2025-01-31","arxiv_id":"2501.19245","repositories_listed":1,"syntology":null},{"url":"/paper/vintix-action-model-via-in-context","slug":"vintix-action-model-via-in-context","title":"Vintix: Action Model via In-Context Reinforcement Learning","date":"2025-01-31","arxiv_id":"2501.19400","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vintix-action-model-via-in-context#ran","syntology_url":"https://syntology.ai/paper/2501.19400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.19400"}},"official":{"repos":["dunnolab/vintix"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/dream-to-drive-with-predictive-individual","slug":"dream-to-drive-with-predictive-individual","title":"Dream to Drive with Predictive Individual World Model","date":"2025-01-28","arxiv_id":"2501.16733","repositories_listed":1,"syntology":null},{"url":"/paper/on-rollouts-in-model-based-reinforcement","slug":"on-rollouts-in-model-based-reinforcement","title":"On Rollouts in Model-Based Reinforcement Learning","date":"2025-01-28","arxiv_id":"2501.16918","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-rollouts-in-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2501.16918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.16918"}},"official":{"repos":["data-science-in-mechanical-engineering/infoprop"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-quantum-reinforcement-learning","slug":"benchmarking-quantum-reinforcement-learning","title":"Benchmarking Quantum Reinforcement Learning","date":"2025-01-27","arxiv_id":"2501.15893","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-quantum-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2501.15893","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.15893"}},"official":{"repos":["nicomeyer96/qrl-benchmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"d038d252b1ba2c53978f113d4f367f86cc2c47c10fa10ed698023c0a07b06734","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}