{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/48","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":48,"pages_in_order":152,"rows_per_page":100,"rows":[4701,4800],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/47","next":"/task/reinforcement-learning-1/papers/49","papers":[{"url":"/paper/playing-doom-with-slam-augmented-deep","slug":"playing-doom-with-slam-augmented-deep","title":"Playing Doom with SLAM-Augmented Deep Reinforcement Learning","date":"2016-12-01","arxiv_id":"1612.00380","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-multi-domain","slug":"deep-reinforcement-learning-for-multi-domain","title":"Deep Reinforcement Learning for Multi-Domain Dialogue Systems","date":"2016-11-26","arxiv_id":"1611.08675","repositories_listed":1,"syntology":null},{"url":"/paper/training-an-interactive-humanoid-robot-using","slug":"training-an-interactive-humanoid-robot-using","title":"Training an Interactive Humanoid Robot Using Multimodal Deep Reinforcement Learning","date":"2016-11-26","arxiv_id":"1611.08666","repositories_listed":1,"syntology":null},{"url":"/paper/a-simple-fast-diverse-decoding-algorithm-for","slug":"a-simple-fast-diverse-decoding-algorithm-for","title":"A Simple, Fast Diverse Decoding Algorithm for Neural Generation","date":"2016-11-25","arxiv_id":"1611.08562","repositories_listed":1,"syntology":null},{"url":"/paper/variational-intrinsic-control","slug":"variational-intrinsic-control","title":"Variational Intrinsic Control","date":"2016-11-22","arxiv_id":"1611.07507","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-object-detection-with-deep","slug":"hierarchical-object-detection-with-deep","title":"Hierarchical Object Detection with Deep Reinforcement Learning","date":"2016-11-11","arxiv_id":"1611.03718","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-navigate-in-complex-environments","slug":"learning-to-navigate-in-complex-environments","title":"Learning to Navigate in Complex Environments","date":"2016-11-11","arxiv_id":"1611.03673","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-play-in-a-day-faster-deep","slug":"learning-to-play-in-a-day-faster-deep","title":"Learning to Play in a Day: Faster Deep Reinforcement Learning by Optimality Tightening","date":"2016-11-05","arxiv_id":"1611.01606","repositories_listed":1,"syntology":null},{"url":"/paper/reset-free-trial-and-error-learning-for-robot","slug":"reset-free-trial-and-error-learning-for-robot","title":"Reset-free Trial-and-Error Learning for Robot Damage Recovery","date":"2016-10-13","arxiv_id":"1610.04213","repositories_listed":1,"syntology":null},{"url":"/paper/active-exploration-in-parameterized","slug":"active-exploration-in-parameterized","title":"Active exploration in parameterized reinforcement learning","date":"2016-10-06","arxiv_id":"1610.01986","repositories_listed":1,"syntology":null},{"url":"/paper/deep-visual-foresight-for-planning-robot","slug":"deep-visual-foresight-for-planning-robot","title":"Deep Visual Foresight for Planning Robot Motion","date":"2016-10-03","arxiv_id":"1610.00696","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-mention","slug":"deep-reinforcement-learning-for-mention","title":"Deep Reinforcement Learning for Mention-Ranking Coreference Models","date":"2016-09-27","arxiv_id":"1609.08667","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/deep-reinforcement-learning-for-mention#ran","syntology_url":"https://syntology.ai/paper/1609.08667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1609.08667"}},"official":{"repos":["clarkkev/deep-coref"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/opponent-modeling-in-deep-reinforcement","slug":"opponent-modeling-in-deep-reinforcement","title":"Opponent Modeling in Deep Reinforcement Learning","date":"2016-09-18","arxiv_id":"1609.05559","repositories_listed":1,"syntology":null},{"url":"/paper/a-threshold-based-scheme-for-reinforcement","slug":"a-threshold-based-scheme-for-reinforcement","title":"A Threshold-based Scheme for Reinforcement Learning in Neural Networks","date":"2016-09-12","arxiv_id":"1609.03348","repositories_listed":1,"syntology":null},{"url":"/paper/towards-end-to-end-reinforcement-learning-of","slug":"towards-end-to-end-reinforcement-learning-of","title":"Towards End-to-End Reinforcement Learning of Dialogue Agents for Information Access","date":"2016-09-03","arxiv_id":"1609.00777","repositories_listed":1,"syntology":null},{"url":"/paper/posterior-sampling-for-reinforcement-learning-1","slug":"posterior-sampling-for-reinforcement-learning-1","title":"Posterior Sampling for Reinforcement Learning Without Episodes","date":"2016-08-09","arxiv_id":"1608.02731","repositories_listed":1,"syntology":null},{"url":"/paper/playing-atari-games-with-deep-reinforcement","slug":"playing-atari-games-with-deep-reinforcement","title":"Playing Atari Games with Deep Reinforcement Learning and Human Checkpoint Replay","date":"2016-07-18","arxiv_id":"1607.05077","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-a","slug":"deep-reinforcement-learning-with-a","title":"Deep Reinforcement Learning with a Combinatorial Action Space for Predicting Popular Reddit Threads","date":"2016-06-12","arxiv_id":"1606.03667","repositories_listed":1,"syntology":null},{"url":"/paper/deep-successor-reinforcement-learning","slug":"deep-successor-reinforcement-learning","title":"Deep Successor Reinforcement Learning","date":"2016-06-08","arxiv_id":"1606.02396","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-successor-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/1606.02396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1606.02396"}},"official":{"repos":["Ardavans/DSR"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-end-to-end-learning-for-dialog-state","slug":"towards-end-to-end-learning-for-dialog-state","title":"Towards End-to-End Learning for Dialog State Tracking and Management using Deep Reinforcement Learning","date":"2016-06-08","arxiv_id":"1606.02560","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-count-based-exploration-and","slug":"unifying-count-based-exploration-and","title":"Unifying Count-Based Exploration and Intrinsic Motivation","date":"2016-06-06","arxiv_id":"1606.01868","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-radio-control-and","slug":"deep-reinforcement-learning-radio-control-and","title":"Deep Reinforcement Learning Radio Control and Signal Detection with KeRLym, a Gym RL Agent","date":"2016-05-30","arxiv_id":"1605.09221","repositories_listed":1,"syntology":null},{"url":"/paper/improving-information-extraction-by-acquiring","slug":"improving-information-extraction-by-acquiring","title":"Improving Information Extraction by Acquiring External Evidence with Reinforcement Learning","date":"2016-03-25","arxiv_id":"1603.07954","repositories_listed":1,"syntology":null},{"url":"/paper/exploratory-gradient-boosting-for","slug":"exploratory-gradient-boosting-for","title":"Exploratory Gradient Boosting for Reinforcement Learning in Complex Domains","date":"2016-03-14","arxiv_id":"1603.04119","repositories_listed":1,"syntology":null},{"url":"/paper/investigating-practical-linear-temporal","slug":"investigating-practical-linear-temporal","title":"Investigating practical linear temporal difference learning","date":"2016-02-28","arxiv_id":"1602.08771","repositories_listed":1,"syntology":null},{"url":"/paper/simpleds-a-simple-deep-reinforcement-learning","slug":"simpleds-a-simple-deep-reinforcement-learning","title":"SimpleDS: A Simple Deep Reinforcement Learning Dialogue System","date":"2016-01-18","arxiv_id":"1601.04574","repositories_listed":1,"syntology":null},{"url":"/paper/angrier-birds-bayesian-reinforcement-learning","slug":"angrier-birds-bayesian-reinforcement-learning","title":"Angrier Birds: Bayesian reinforcement learning","date":"2016-01-06","arxiv_id":"1601.01297","repositories_listed":1,"syntology":null},{"url":"/paper/state-of-the-art-control-of-atari-games-using","slug":"state-of-the-art-control-of-atari-games-using","title":"State of the Art Control of Atari Games Using Shallow Reinforcement Learning","date":"2015-12-04","arxiv_id":"1512.01563","repositories_listed":1,"syntology":null},{"url":"/paper/strategic-dialogue-management-via-deep","slug":"strategic-dialogue-management-via-deep","title":"Strategic Dialogue Management via Deep Reinforcement Learning","date":"2015-11-25","arxiv_id":"1511.08099","repositories_listed":1,"syntology":null},{"url":"/paper/conditional-computation-in-neural-networks","slug":"conditional-computation-in-neural-networks","title":"Conditional Computation in Neural Networks for faster models","date":"2015-11-19","arxiv_id":"1511.06297","repositories_listed":1,"syntology":null},{"url":"/paper/policy-distillation","slug":"policy-distillation","title":"Policy Distillation","date":"2015-11-19","arxiv_id":"1511.06295","repositories_listed":1,"syntology":null},{"url":"/paper/deep-spatial-autoencoders-for-visuomotor","slug":"deep-spatial-autoencoders-for-visuomotor","title":"Deep Spatial Autoencoders for Visuomotor Learning","date":"2015-09-21","arxiv_id":"1509.06113","repositories_listed":1,"syntology":null},{"url":"/paper/action-conditional-video-prediction-using","slug":"action-conditional-video-prediction-using","title":"Action-Conditional Video Prediction using Deep Networks in Atari Games","date":"2015-07-31","arxiv_id":"1507.08750","repositories_listed":1,"syntology":null},{"url":"/paper/incentivizing-exploration-in-reinforcement","slug":"incentivizing-exploration-in-reinforcement","title":"Incentivizing Exploration In Reinforcement Learning With Deep Predictive Models","date":"2015-07-03","arxiv_id":"1507.00814","repositories_listed":1,"syntology":null},{"url":"/paper/learning-where-to-sample-in-structured","slug":"learning-where-to-sample-in-structured","title":"Learning Where to Sample in Structured Prediction","date":"2015-05-09","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-neural-turing-machines","slug":"reinforcement-learning-neural-turing-machines","title":"Reinforcement Learning Neural Turing Machines - Revised","date":"2015-05-04","arxiv_id":"1505.00521","repositories_listed":1,"syntology":null},{"url":"/paper/gaussian-processes-for-data-efficient","slug":"gaussian-processes-for-data-efficient","title":"Gaussian Processes for Data-Efficient Learning in Robotics and Control","date":"2015-02-10","arxiv_id":"1502.02860","repositories_listed":1,"syntology":null},{"url":"/paper/deterministic-policy-gradient-algorithms","slug":"deterministic-policy-gradient-algorithms","title":"Deterministic Policy Gradient Algorithms","date":"2014-06-22","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-learning-in-neural-networks-an-overview","slug":"deep-learning-in-neural-networks-an-overview","title":"Deep Learning in Neural Networks: An Overview","date":"2014-04-30","arxiv_id":"1404.7828","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-the-cvar-via-sampling","slug":"optimizing-the-cvar-via-sampling","title":"Optimizing the CVaR via Sampling","date":"2014-04-15","arxiv_id":"1404.3862","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-planning-and-learning-for-multiagent","slug":"scalable-planning-and-learning-for-multiagent","title":"Scalable Planning and Learning for Multiagent POMDPs: Extended Version","date":"2014-04-04","arxiv_id":"1404.1140","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-general-value-functions-to","slug":"off-policy-general-value-functions-to","title":"Off-Policy General Value Functions to Represent Dynamic Role Assignments in RoboCup 3D Soccer Simulation","date":"2014-02-18","arxiv_id":"1402.4525","repositories_listed":1,"syntology":null},{"url":"/paper/generalization-and-exploration-via-randomized","slug":"generalization-and-exploration-via-randomized","title":"Generalization and Exploration via Randomized Value Functions","date":"2014-02-04","arxiv_id":"1402.0635","repositories_listed":1,"syntology":null},{"url":"/paper/using-reinforcement-learning-to-find-an","slug":"using-reinforcement-learning-to-find-an","title":"Using reinforcement learning to find an optimal set of features","date":"2013-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/artist-agent-a-reinforcement-learning","slug":"artist-agent-a-reinforcement-learning","title":"Artist Agent: A Reinforcement Learning Approach to Automatic Stroke Generation in Oriental Ink Painting","date":"2012-06-18","arxiv_id":"1206.4634","repositories_listed":1,"syntology":null},{"url":"/paper/off-policy-actor-critic","slug":"off-policy-actor-critic","title":"Off-Policy Actor-Critic","date":"2012-05-22","arxiv_id":"1205.4839","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/off-policy-actor-critic#ran","syntology_url":"https://syntology.ai/paper/1205.4839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1205.4839"}},"official":null}},{"url":"/paper/nonlinear-inverse-reinforcement-learning-with","slug":"nonlinear-inverse-reinforcement-learning-with","title":"Nonlinear Inverse Reinforcement Learning with Gaussian Processes","date":"2011-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/an-object-oriented-representation-for","slug":"an-object-oriented-representation-for","title":"An Object-Oriented Representation for Efficient Reinforcement Learning","date":"2008-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/least-squares-policy-iteration","slug":"least-squares-policy-iteration","title":"Least-Squares Policy Iteration","date":"2003-12-04","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":null,"slug":"aligning-humans-and-robots-via-reinforcement","title":"Aligning Humans and Robots via Reinforcement Learning from Implicit Human Feedback","date":"2025-07-17","arxiv_id":"2507.13171","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-novelty-to-imitation-self-distilled","title":"From Novelty to Imitation: Self-Distilled Rewards for Offline Reinforcement Learning","date":"2025-07-17","arxiv_id":"2507.12815","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-meets-large","title":"Inverse Reinforcement Learning Meets Large Language Model Post-Training: Basics, Advances, and Opportunities","date":"2025-07-17","arxiv_id":"2507.13158","repositories_listed":0,"syntology":null},{"url":"/paper/questa-expanding-reasoning-capacity-in-llms","slug":"questa-expanding-reasoning-capacity-in-llms","title":"QuestA: Expanding Reasoning Capacity in LLMs via Question Augmentation","date":"2025-07-17","arxiv_id":"2507.13266","repositories_listed":0,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":12,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/questa-expanding-reasoning-capacity-in-llms#ran","syntology_url":"https://syntology.ai/paper/2507.13266","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.13266"}},"official":null}},{"url":null,"slug":"supervised-fine-tuning-on-curated-data-is","title":"Supervised Fine Tuning on Curated Data is Reinforcement Learning (and can be improved)","date":"2025-07-17","arxiv_id":"2507.12856","repositories_listed":0,"syntology":null},{"url":null,"slug":"var-math-probing-true-mathematical-reasoning","title":"VAR-MATH: Probing True Mathematical Reasoning in Large Language Models via Symbolic Multi-Instance Benchmarks","date":"2025-07-17","arxiv_id":"2507.12885","repositories_listed":0,"syntology":null},{"url":null,"slug":"fly-fail-fix-iterative-game-repair-with","title":"Fly, Fail, Fix: Iterative Game Repair with Reinforcement Learning and Large Multimodal Models","date":"2025-07-16","arxiv_id":"2507.12666","repositories_listed":0,"syntology":null},{"url":null,"slug":"kevin-multi-turn-rl-for-generating-cuda","title":"Kevin: Multi-Turn RL for Generating CUDA Kernels","date":"2025-07-16","arxiv_id":"2507.11948","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-up-rl-unlocking-diverse-reasoning-in","title":"Scaling Up RL: Unlocking Diverse Reasoning in LLMs via Prolonged Training","date":"2025-07-16","arxiv_id":"2507.12507","repositories_listed":0,"syntology":null},{"url":null,"slug":"illuminating-the-three-dogmas-of","title":"Illuminating the Three Dogmas of Reinforcement Learning under Evolutionary Light","date":"2025-07-15","arxiv_id":"2507.11482","repositories_listed":0,"syntology":null},{"url":null,"slug":"local-pairwise-distance-matching-for","title":"Local Pairwise Distance Matching for Backpropagation-Free Reinforcement Learning","date":"2025-07-15","arxiv_id":"2507.11367","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-bayesian-detection-of-drift-evasive","title":"Real-Time Bayesian Detection of Drift-Evasive GNSS Spoofing in Reinforcement Learning Based UAV Deconfliction","date":"2025-07-15","arxiv_id":"2507.11173","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-synergy-dilemma-of-long-cot-sft-and-rl","title":"The Synergy Dilemma of Long-CoT SFT and RL: Investigating Post-Training Techniques for Reasoning VLMs","date":"2025-07-10","arxiv_id":"2507.07562","repositories_listed":0,"syntology":null},{"url":null,"slug":"squeeze-the-soaked-sponge-efficient-off","title":"Squeeze the Soaked Sponge: Efficient Off-policy Reinforcement Finetuning for Large Language Model","date":"2025-07-09","arxiv_id":"2507.06892","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-rts-rethinking-reinforcement-learning","title":"Video-RTS: Rethinking Reinforcement Learning and Test-Time Scaling for Efficient and Enhanced Video Reasoning","date":"2025-07-09","arxiv_id":"2507.06485","repositories_listed":0,"syntology":null},{"url":null,"slug":"cognisql-r1-zero-lightweight-reinforced","title":"CogniSQL-R1-Zero: Lightweight Reinforced Reasoning for Efficient SQL Generation","date":"2025-07-08","arxiv_id":"2507.06013","repositories_listed":0,"syntology":null},{"url":null,"slug":"detecting-and-mitigating-reward-hacking-in","title":"Detecting and Mitigating Reward Hacking in Reinforcement Learning Systems: A Comprehensive Empirical Study","date":"2025-07-08","arxiv_id":"2507.05619","repositories_listed":0,"syntology":null},{"url":null,"slug":"fevo-financial-knowledge-expansion-and","title":"FEVO: Financial Knowledge Expansion and Reasoning Evolution for Large Language Models","date":"2025-07-08","arxiv_id":"2507.06057","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-bandwidth-estimation-for-real-time","title":"Robust Bandwidth Estimation for Real-Time Communication with Offline Reinforcement Learning","date":"2025-07-08","arxiv_id":"2507.05785","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-domain-randomization-via-uncertainty","title":"Safe Domain Randomization via Uncertainty-Aware Out-of-Distribution Detection and Policy Adaptation","date":"2025-07-08","arxiv_id":"2507.06111","repositories_listed":0,"syntology":null},{"url":null,"slug":"2048-reinforcement-learning-in-a-delayed","title":"2048: Reinforcement Learning in a Delayed Reward Environment","date":"2025-07-07","arxiv_id":"2507.05465","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-vision-reasoner-transferring-linguistic","title":"Open Vision Reasoner: Transferring Linguistic Cognitive Behavior for Visual Reasoning","date":"2025-07-07","arxiv_id":"2507.05255","repositories_listed":0,"syntology":null},{"url":null,"slug":"listener-rewarded-thinking-in-vlms-for-image","title":"Listener-Rewarded Thinking in VLMs for Image Preferences","date":"2025-06-28","arxiv_id":"2506.22832","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-continual-reinforcement-learning","title":"A Survey of Continual Reinforcement Learning","date":"2025-06-27","arxiv_id":"2506.21872","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancements-and-challenges-in-continual","title":"Advancements and Challenges in Continual Reinforcement Learning: A Comprehensive Review","date":"2025-06-27","arxiv_id":"2506.21899","repositories_listed":0,"syntology":null},{"url":null,"slug":"apo-enhancing-reasoning-ability-of-mllms-via","title":"APO: Enhancing Reasoning Ability of MLLMs via Asymmetric Policy Optimization","date":"2025-06-26","arxiv_id":"2506.21655","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-guided-antifragile-reinforcement","title":"Curriculum-Guided Antifragile Reinforcement Learning for Secure UAV Deconfliction under Observation-Space Attacks","date":"2025-06-26","arxiv_id":"2506.21129","repositories_listed":0,"syntology":null},{"url":"/paper/flow-based-single-step-completion-for","slug":"flow-based-single-step-completion-for","title":"Flow-Based Single-Step Completion for Efficient and Expressive Policy Learning","date":"2025-06-26","arxiv_id":"2506.21427","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/flow-based-single-step-completion-for#ran","syntology_url":"https://syntology.ai/paper/2506.21427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.21427"}},"official":null}},{"url":null,"slug":"optimising-4th-order-runge-kutta-methods-a","title":"Optimising 4th-Order Runge-Kutta Methods: A Dynamic Heuristic Approach for Efficiency and Low Storage","date":"2025-06-26","arxiv_id":"2506.21465","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-selector-reinforcement-learning-guided","title":"RL-Selector: Reinforcement Learning-Guided Data Selection via Redundancy Assessment","date":"2025-06-26","arxiv_id":"2506.21037","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-policy-switching-for-antifragile","title":"Robust Policy Switching for Antifragile Reinforcement Learning for UAV Deconfliction in Adversarial Environments","date":"2025-06-26","arxiv_id":"2506.21127","repositories_listed":0,"syntology":null},{"url":null,"slug":"strict-subgoal-execution-reliable-long","title":"Strict Subgoal Execution: Reliable Long-Horizon Planning in Hierarchical Reinforcement Learning","date":"2025-06-26","arxiv_id":"2506.21039","repositories_listed":0,"syntology":null},{"url":null,"slug":"asymmetric-reinforce-for-off-policy","title":"Asymmetric REINFORCE for off-Policy Reinforcement Learning: Balancing positive and negative rewards","date":"2025-06-25","arxiv_id":"2506.20520","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparative-analysis-of-reinforcement","title":"A Comparative Analysis of Reinforcement Learning and Conventional Deep Learning Approaches for Bearing Fault Diagnosis","date":"2025-06-24","arxiv_id":"2506.19929","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-aware-intelligent-qoe-optimization-for","title":"Causal-Aware Intelligent QoE Optimization for VR Interaction with Adaptive Keyframe Extraction","date":"2025-06-24","arxiv_id":"2506.19890","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-and-value","title":"Hierarchical Reinforcement Learning and Value Optimization for Challenging Quadruped Locomotion","date":"2025-06-24","arxiv_id":"2506.20036","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapthink-adaptive-thinking-preferences-for","title":"AdapThink: Adaptive Thinking Preferences for Reasoning Language Model","date":"2025-06-23","arxiv_id":"2506.18237","repositories_listed":0,"syntology":null},{"url":null,"slug":"robots-and-children-that-learn-together","title":"Robots and Children that Learn Together : Improving Knowledge Retention by Teaching Peer-Like Interactive Robots","date":"2025-06-23","arxiv_id":"2506.18365","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-residual-reinforcement-learning","title":"Accelerating Residual Reinforcement Learning with Uncertainty Estimation","date":"2025-06-21","arxiv_id":"2506.17564","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveling-the-playing-field-carefully","title":"Leveling the Playing Field: Carefully Comparing Classical and Learned Controllers for Quadrotor Trajectory Tracking","date":"2025-06-21","arxiv_id":"2506.17832","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-dexterous-object-handover","title":"Learning Dexterous Object Handover","date":"2025-06-20","arxiv_id":"2506.16822","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-objective-reinforcement-learning-with","title":"Dual-Objective Reinforcement Learning with Novel Hamilton-Jacobi-Bellman Formulations","date":"2025-06-19","arxiv_id":"2506.16016","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-general-to-targeted-rewards-surpassing","title":"From General to Targeted Rewards: Surpassing GPT-4 in Open-Ended Long-Context Generation","date":"2025-06-19","arxiv_id":"2506.16024","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-lifelong-reinforcement-learning","title":"Multi-Task Lifelong Reinforcement Learning for Wireless Sensor Networks","date":"2025-06-19","arxiv_id":"2506.16254","repositories_listed":0,"syntology":null},{"url":null,"slug":"vrail-vectorized-reward-based-attribution-for","title":"VRAIL: Vectorized Reward-based Attribution for Interpretable Learning","date":"2025-06-19","arxiv_id":"2506.16014","repositories_listed":0,"syntology":null},{"url":null,"slug":"make-your-auv-adaptive-an-environment-aware","title":"Make Your AUV Adaptive: An Environment-Aware Reinforcement Learning Framework For Underwater Tasks","date":"2025-06-18","arxiv_id":"2506.15082","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-28","title":"Multi-Agent Reinforcement Learning for Autonomous Multi-Satellite Earth Observation: A Realistic Case Study","date":"2025-06-18","arxiv_id":"2506.15207","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-policy","title":"Reinforcement Learning-Based Policy Optimisation For Heterogeneous Radio Access","date":"2025-06-18","arxiv_id":"2506.15273","repositories_listed":0,"syntology":null},{"url":null,"slug":"steering-your-diffusion-policy-with-latent","title":"Steering Your Diffusion Policy with Latent Space Reinforcement Learning","date":"2025-06-18","arxiv_id":"2506.15799","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-reinforcement-learning-for","title":"Adaptive Reinforcement Learning for Unobservable Random Delays","date":"2025-06-17","arxiv_id":"2506.14411","repositories_listed":0,"syntology":null},{"url":null,"slug":"hilight-a-hierarchical-reinforcement-learning","title":"HiLight: A Hierarchical Reinforcement Learning Framework with Global Adversarial Guidance for Large-Scale Traffic Signal Control","date":"2025-06-17","arxiv_id":"2506.14391","repositories_listed":0,"syntology":null}],"record_sha256":"551c0e35f54164d66f58c6c13241ba1ab86482ca2816d534c27487614f866aba","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}