{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/54","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":54,"pages_in_order":132,"rows_per_page":100,"rows":[5301,5400],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/53","next":"/task/reinforcement-learning/papers/55","papers":[{"url":null,"slug":"optimizing-deep-reinforcement-learning-for-1","title":"Optimizing Deep Reinforcement Learning for Adaptive Robotic Arm Control","date":"2024-06-12","arxiv_id":"2407.02503","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-high-level","title":"Reinforcement Learning for High-Level Strategic Control in Tower Defense Games","date":"2024-06-12","arxiv_id":"2406.07980","repositories_listed":0,"syntology":null},{"url":null,"slug":"rile-reinforced-imitation-learning","title":"RILe: Reinforced Imitation Learning","date":"2024-06-12","arxiv_id":"2406.08472","repositories_listed":0,"syntology":null},{"url":null,"slug":"time-constrained-robust-mdps","title":"Time-Constrained Robust MDPs","date":"2024-06-12","arxiv_id":"2406.08395","repositories_listed":0,"syntology":null},{"url":null,"slug":"cdsa-conservative-denoising-score-based","title":"CDSA: Conservative Denoising Score-based Algorithm for Offline Reinforcement Learning","date":"2024-06-11","arxiv_id":"2406.07541","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-reinforcement-learning-from-offline","title":"Hybrid Reinforcement Learning from Offline Observation Alone","date":"2024-06-11","arxiv_id":"2406.07253","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-demonstration-and-preference-learning","title":"Learning Reward and Policy Jointly from Demonstration and Preference Improves Alignment","date":"2024-06-11","arxiv_id":"2406.06874","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-robustness-in-preference-based","title":"Boosting Robustness in Preference-Based Reinforcement Learning with Dynamic Sparsity","date":"2024-06-10","arxiv_id":"2406.06495","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitivity-in-markov-games-and-multi","title":"Risk Sensitivity in Markov Games and Multi-Agent Reinforcement Learning: A Systematic Review","date":"2024-06-10","arxiv_id":"2406.06041","repositories_listed":0,"syntology":null},{"url":null,"slug":"verification-guided-shielding-for-deep","title":"Verification-Guided Shielding for Deep Reinforcement Learning","date":"2024-06-10","arxiv_id":"2406.06507","repositories_listed":0,"syntology":null},{"url":null,"slug":"lgr2-language-guided-reward-relabeling-for","title":"LGR2: Language Guided Reward Relabeling for Accelerating Hierarchical Reinforcement Learning","date":"2024-06-09","arxiv_id":"2406.05881","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-flight-envelope-protection-a-novel","title":"Enhanced Flight Envelope Protection: A Novel Reinforcement Learning Approach","date":"2024-06-08","arxiv_id":"2406.05586","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-intensity-control","title":"Reinforcement Learning for Intensity Control: An Application to Choice-Based Network Revenue Management","date":"2024-06-08","arxiv_id":"2406.05358","repositories_listed":0,"syntology":null},{"url":"/paper/optimizing-automatic-differentiation-with","slug":"optimizing-automatic-differentiation-with","title":"Optimizing Automatic Differentiation with Deep Reinforcement Learning","date":"2024-06-07","arxiv_id":"2406.05027","repositories_listed":0,"syntology":{"n":29,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":17,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 17 unverified","sample_list":"/paper/optimizing-automatic-differentiation-with#ran","syntology_url":"https://syntology.ai/paper/2406.05027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05027"}},"official":null}},{"url":null,"slug":"ac4mpc-actor-critic-reinforcement-learning","title":"AC4MPC: Actor-Critic Reinforcement Learning for Nonlinear Model Predictive Control","date":"2024-06-06","arxiv_id":"2406.03995","repositories_listed":0,"syntology":null},{"url":null,"slug":"atradiff-accelerating-online-reinforcement","title":"ATraDiff: Accelerating Online Reinforcement Learning with Imaginary Trajectories","date":"2024-06-06","arxiv_id":"2406.04323","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavior-targeted-attack-on-reinforcement","title":"Behavior-Targeted Attack on Reinforcement Learning with Limited Access to Victim's Policy","date":"2024-06-06","arxiv_id":"2406.03862","repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrapping-expectiles-in-reinforcement","title":"Bootstrapping Expectiles in Reinforcement Learning","date":"2024-06-06","arxiv_id":"2406.04081","repositories_listed":0,"syntology":null},{"url":null,"slug":"breeding-programs-optimization-with","title":"Breeding Programs Optimization with Reinforcement Learning","date":"2024-06-06","arxiv_id":"2406.03932","repositories_listed":0,"syntology":null},{"url":null,"slug":"excluding-the-irrelevant-focusing","title":"Excluding the Irrelevant: Focusing Reinforcement Learning through Continuous Action Masking","date":"2024-06-06","arxiv_id":"2406.03704","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-pessimism-and-optimism-dynamics-in","title":"Exploring Pessimism and Optimism Dynamics in Deep Reinforcement Learning","date":"2024-06-06","arxiv_id":"2406.03890","repositories_listed":0,"syntology":null},{"url":null,"slug":"gensafe-a-generalizable-safety-enhancer-for","title":"GenSafe: A Generalizable Safety Enhancer for Safe Reinforcement Learning Algorithms Based on Reduced Order Markov Decision Process Model","date":"2024-06-06","arxiv_id":"2406.03912","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-autonomous-driving-for-safety-a","title":"Optimizing Autonomous Driving for Safety: A Human-Centric Approach with LLM-Enhanced RLHF","date":"2024-06-06","arxiv_id":"2406.04481","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-play-with-adversarial-critic-provable","title":"Self-Play with Adversarial Critic: Provable and Scalable Offline Alignment for Language Models","date":"2024-06-06","arxiv_id":"2406.04274","repositories_listed":0,"syntology":null},{"url":null,"slug":"inductive-generalization-in-reinforcement","title":"Inductive Generalization in Reinforcement Learning from Specifications","date":"2024-06-05","arxiv_id":"2406.03651","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-informed-critic-in-an-actor-critic","title":"Physics-Guided Actor-Critic Reinforcement Learning for Swimming in Turbulence","date":"2024-06-05","arxiv_id":"2406.10242","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-learning-for-efficient-deep","title":"Representation Learning For Efficient Deep Multi-Agent Reinforcement Learning","date":"2024-06-05","arxiv_id":"2406.02890","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unifying-framework-for-action-conditional","title":"A Unifying Framework for Action-Conditional Self-Predictive Reinforcement Learning","date":"2024-06-04","arxiv_id":"2406.02035","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-preference-scaling-for-reinforcement","title":"Adaptive Preference Scaling for Reinforcement Learning with Human Feedback","date":"2024-06-04","arxiv_id":"2406.02764","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithmic-collusion-in-dynamic-pricing-with","title":"Algorithmic Collusion in Dynamic Pricing with Deep Reinforcement Learning","date":"2024-06-04","arxiv_id":"2406.02437","repositories_listed":0,"syntology":null},{"url":null,"slug":"fightladder-a-benchmark-for-competitive-multi","title":"FightLadder: A Benchmark for Competitive Multi-Agent Reinforcement Learning","date":"2024-06-04","arxiv_id":"2406.02081","repositories_listed":0,"syntology":null},{"url":null,"slug":"iqrl-implicitly-quantized-representations-for","title":"iQRL -- Implicitly Quantized Representations for Sample-efficient Reinforcement Learning","date":"2024-06-04","arxiv_id":"2406.02696","repositories_listed":0,"syntology":null},{"url":null,"slug":"rectifying-reinforcement-learning-for-reward","title":"Rectifying Reinforcement Learning for Reward Matching","date":"2024-06-04","arxiv_id":"2406.02213","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-architecture","title":"Reinforcement learning-based architecture search for quantum machine learning","date":"2024-06-04","arxiv_id":"2406.02717","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-lookahead","title":"Reinforcement Learning with Lookahead Information","date":"2024-06-04","arxiv_id":"2406.02258","repositories_listed":0,"syntology":null},{"url":null,"slug":"test-time-regret-minimization-in-meta","title":"Test-Time Regret Minimization in Meta Reinforcement Learning","date":"2024-06-04","arxiv_id":"2406.02282","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-view-on-planning-in-online","title":"A New View on Planning in Online Reinforcement Learning","date":"2024-06-03","arxiv_id":"2406.01562","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-prompting-model-based-offline","title":"Causal prompting model-based offline reinforcement learning","date":"2024-06-03","arxiv_id":"2406.01065","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-behavioral-mode","title":"Deep Reinforcement Learning Behavioral Mode Switching Using Optimal Control Based on a Latent Space Objective","date":"2024-06-03","arxiv_id":"2406.01178","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-weakly","title":"Deep reinforcement learning for weakly coupled MDP's with continuous actions","date":"2024-06-03","arxiv_id":"2406.01099","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-assignment-via-state-augmented","title":"Multi-agent assignment via state augmented reinforcement learning","date":"2024-06-03","arxiv_id":"2406.01782","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-meets-leaf","title":"Multi-Agent Reinforcement Learning Meets Leaf Sequencing in Radiotherapy","date":"2024-06-03","arxiv_id":"2406.01853","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-ensembling-for-mitigating-reward","title":"Scalable Ensembling For Mitigating Reward Overoptimisation","date":"2024-06-03","arxiv_id":"2406.01013","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximum-entropy-regularized-decision","title":"Maximum-Entropy Regularized Decision Transformer with Reward Relabelling for Dynamic Recommendation","date":"2024-06-02","arxiv_id":"2406.00725","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-dimensional-optimization-for-text","title":"Multi-Dimensional Optimization for Text Summarization via Reinforcement Learning","date":"2024-06-01","arxiv_id":"2406.00303","repositories_listed":0,"syntology":null},{"url":null,"slug":"decision-mamba-reinforcement-learning-via-1","title":"Decision Mamba: Reinforcement Learning via Hybrid Selective Sequence Modeling","date":"2024-05-31","arxiv_id":"2406.00079","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploratory-preference-optimization","title":"Exploratory Preference Optimization: Harnessing Implicit Q*-Approximation for Sample-Efficient RLHF","date":"2024-05-31","arxiv_id":"2405.21046","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-sociohydrology","title":"Reinforcement Learning for Sociohydrology","date":"2024-05-31","arxiv_id":"2405.20772","repositories_listed":0,"syntology":null},{"url":null,"slug":"bilevel-reinforcement-learning-via-the","title":"Bilevel reinforcement learning via the development of hyper-gradient without lower-level convexity","date":"2024-05-30","arxiv_id":"2405.19697","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-feature-selection-in-medical","title":"Dynamic feature selection in medical predictive monitoring by reinforcement learning","date":"2024-05-30","arxiv_id":"2405.19729","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-stimuli-generation-using","title":"Efficient Stimuli Generation using Reinforcement Learning in Design Verification","date":"2024-05-30","arxiv_id":"2405.19815","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-reinforcement-learning-framework-for","title":"Hybrid Reinforcement Learning Framework for Mixed-Variable Problems","date":"2024-05-30","arxiv_id":"2405.20500","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-random-demonstrations-offline","title":"Learning from Random Demonstrations: Offline Reinforcement Learning with Importance-Sampled Diffusion Models","date":"2024-05-30","arxiv_id":"2405.19878","repositories_listed":0,"syntology":null},{"url":null,"slug":"metacurl-non-stationary-concave-utility","title":"MetaCURL: Non-stationary Concave Utility Reinforcement Learning","date":"2024-05-30","arxiv_id":"2405.19807","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-as-a-monotone-scheme","title":"Q-learning as a monotone scheme","date":"2024-05-30","arxiv_id":"2405.20538","repositories_listed":0,"syntology":null},{"url":null,"slug":"randomized-exploration-for-reinforcement-1","title":"Randomized Exploration for Reinforcement Learning with Multinomial Logistic Function Approximation","date":"2024-05-30","arxiv_id":"2405.20165","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-multi-agent-reinforcement-learning-with-1","title":"Safe Multi-agent Reinforcement Learning with Natural Language Constraints","date":"2024-05-30","arxiv_id":"2405.20018","repositories_listed":0,"syntology":null},{"url":null,"slug":"sleepernets-universal-backdoor-poisoning","title":"SleeperNets: Universal Backdoor Poisoning Attacks Against Reinforcement Learning Agents","date":"2024-05-30","arxiv_id":"2405.20539","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-household-robotics-deep-interactive","title":"Advancing Household Robotics: Deep Interactive Reinforcement Learning for Efficient Training and Enhanced Performance","date":"2024-05-29","arxiv_id":"2405.18687","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-preference-based-reinforcement-1","title":"Efficient Preference-based Reinforcement Learning via Aligned Experience Estimation","date":"2024-05-29","arxiv_id":"2405.18688","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-concave-utility-reinforcement","title":"Inverse Concave-Utility Reinforcement Learning is Inverse Game Theory","date":"2024-05-29","arxiv_id":"2405.19024","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-regularised-reinforcement-learning","title":"Offline Regularised Reinforcement Learning for Large Language Models Alignment","date":"2024-05-29","arxiv_id":"2405.19107","repositories_listed":0,"syntology":null},{"url":null,"slug":"preferred-action-optimized-diffusion-policies","title":"Preferred-Action-Optimized Diffusion Policies for Offline Reinforcement Learning","date":"2024-05-29","arxiv_id":"2405.18729","repositories_listed":0,"syntology":null},{"url":null,"slug":"rich-observation-reinforcement-learning-with","title":"Rich-Observation Reinforcement Learning with Continuous Latent Dynamics","date":"2024-05-29","arxiv_id":"2405.19269","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectral-risk-safe-reinforcement-learning","title":"Spectral-Risk Safe Reinforcement Learning with Convergence Guarantees","date":"2024-05-29","arxiv_id":"2405.18698","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-incentivized-preference-optimization-a","title":"Value-Incentivized Preference Optimization: A Unified Approach to Online and Offline RLHF","date":"2024-05-29","arxiv_id":"2405.19320","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-reinforcement-learning-in-energy-systems","title":"Why Reinforcement Learning in Energy Systems Needs Explanations","date":"2024-05-29","arxiv_id":"2405.18823","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-pontryagin-perspective-on-reinforcement","title":"A Pontryagin Perspective on Reinforcement Learning","date":"2024-05-28","arxiv_id":"2405.18100","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-horizon-actor-critic-for-policy","title":"Adaptive Horizon Actor-Critic for Policy Learning in Contact-Rich Differentiable Simulation","date":"2024-05-28","arxiv_id":"2405.17784","repositories_listed":0,"syntology":null},{"url":null,"slug":"highway-reinforcement-learning","title":"Highway Reinforcement Learning","date":"2024-05-28","arxiv_id":"2405.18289","repositories_listed":0,"syntology":null},{"url":null,"slug":"mutation-bias-learning-in-games","title":"Mutation-Bias Learning in Games","date":"2024-05-28","arxiv_id":"2405.18190","repositories_listed":0,"syntology":null},{"url":null,"slug":"biological-neurons-compete-with-deep","title":"Biological Neurons Compete with Deep Reinforcement Learning in Sample Efficiency in a Simulated Gameworld","date":"2024-05-27","arxiv_id":"2405.16946","repositories_listed":0,"syntology":null},{"url":null,"slug":"opinion-guided-reinforcement-learning","title":"Opinion-Guided Reinforcement Learning","date":"2024-05-27","arxiv_id":"2405.17287","repositories_listed":0,"syntology":null},{"url":null,"slug":"oracle-efficient-reinforcement-learning-for","title":"Oracle-Efficient Reinforcement Learning for Max Value Ensembles","date":"2024-05-27","arxiv_id":"2405.16739","repositories_listed":0,"syntology":null},{"url":null,"slug":"partial-models-for-building-adaptive-model","title":"Partial Models for Building Adaptive Model-Based Reinforcement Learning Agents","date":"2024-05-27","arxiv_id":"2405.16899","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-reinforcement-learning-5","title":"Provably Efficient Reinforcement Learning with Multinomial Logit Function Approximation","date":"2024-05-27","arxiv_id":"2405.17061","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-escape-route","title":"Reinforcement Learning Based Escape Route Generation in Low Visibility Environments","date":"2024-05-27","arxiv_id":"2406.07568","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-cmdp-within-online-framework-for-meta-safe","title":"A CMDP-within-online framework for Meta-Safe Reinforcement Learning","date":"2024-05-26","arxiv_id":"2405.16601","repositories_listed":0,"syntology":null},{"url":null,"slug":"amortized-active-causal-induction-with-deep","title":"Amortized Active Causal Induction with Deep Reinforcement Learning","date":"2024-05-26","arxiv_id":"2405.16718","repositories_listed":0,"syntology":null},{"url":null,"slug":"make-safe-decisions-in-power-system-safe","title":"Make Safe Decisions in Power System: Safe Reinforcement Learning Based Pre-decision Making for Voltage Stability Emergency Control","date":"2024-05-26","arxiv_id":"2405.16485","repositories_listed":0,"syntology":null},{"url":null,"slug":"pick-up-the-pace-a-parameter-free-optimizer","title":"Fast TRAC: A Parameter-Free Optimizer for Lifelong Reinforcement Learning","date":"2024-05-26","arxiv_id":"2405.16642","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-jump-diffusions","title":"Reinforcement Learning for Jump-Diffusions, with Financial Applications","date":"2024-05-26","arxiv_id":"2405.16449","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlsf-reinforcement-learning-via-symbolic","title":"RLSF: Reinforcement Learning via Symbolic Feedback","date":"2024-05-26","arxiv_id":"2405.16661","repositories_listed":0,"syntology":null},{"url":"/paper/adaptive-q-network-on-the-fly-target","slug":"adaptive-q-network-on-the-fly-target","title":"Adaptive $Q$-Network: On-the-fly Target Selection for Deep Reinforcement Learning","date":"2024-05-25","arxiv_id":"2405.16195","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/adaptive-q-network-on-the-fly-target#ran","syntology_url":"https://syntology.ai/paper/2405.16195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16195"}},"official":null}},{"url":null,"slug":"dynamic-inhomogeneous-quantum-resource","title":"Dynamic Inhomogeneous Quantum Resource Scheduling with Reinforcement Learning","date":"2024-05-25","arxiv_id":"2405.16380","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-analysis-for-conflict-avoidant","title":"Theoretical Study of Conflict-Avoidant Multi-Objective Reinforcement Learning","date":"2024-05-25","arxiv_id":"2405.16077","repositories_listed":0,"syntology":null},{"url":null,"slug":"cooperative-backdoor-attack-in-decentralized","title":"Cooperative Backdoor Attack in Decentralized Reinforcement Learning with Theoretical Guarantee","date":"2024-05-24","arxiv_id":"2405.15245","repositories_listed":0,"syntology":null},{"url":null,"slug":"counterexample-guided-repair-of-reinforcement","title":"Counterexample-Guided Repair of Reinforcement Learning Systems Using Safety Critics","date":"2024-05-24","arxiv_id":"2405.15430","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforcement-learning-via-large","title":"Extracting Heuristics from Large Language Models for Reward Shaping in Reinforcement Learning","date":"2024-05-24","arxiv_id":"2405.15194","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-rlignment-inverse-reinforcement","title":"Inverse-RLignment: Large Language Model Alignment from Demonstrations through Inverse Reinforcement Learning","date":"2024-05-24","arxiv_id":"2405.15624","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-generalizable-human-motion-generator","title":"Learning Generalizable Human Motion Generator with Reinforcement Learning","date":"2024-05-24","arxiv_id":"2405.15541","repositories_listed":0,"syntology":null},{"url":null,"slug":"momentum-based-federated-reinforcement","title":"Momentum-Based Federated Reinforcement Learning with Interaction and Communication Efficiency","date":"2024-05-24","arxiv_id":"2405.17471","repositories_listed":0,"syntology":null},{"url":null,"slug":"trojanforge-adversarial-hardware-trojan","title":"TrojanForge: Generating Adversarial Hardware Trojan Examples Using Reinforcement Learning","date":"2024-05-24","arxiv_id":"2405.15184","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-finite-time-analysis-of-distributed-q","title":"A finite time analysis of distributed Q-learning","date":"2024-05-23","arxiv_id":"2405.14078","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-5-5","title":"Deep Reinforcement Learning for 5*5 Multiplayer Go","date":"2024-05-23","arxiv_id":"2405.14265","repositories_listed":0,"syntology":null},{"url":null,"slug":"deterministic-policies-for-constrained","title":"Deterministic Policies for Constrained Reinforcement Learning in Polynomial Time","date":"2024-05-23","arxiv_id":"2405.14183","repositories_listed":0,"syntology":null},{"url":null,"slug":"direct-preference-optimization-with","title":"Direct Preference Optimization With Unobserved Preference Heterogeneity","date":"2024-05-23","arxiv_id":"2405.15065","repositories_listed":0,"syntology":null},{"url":null,"slug":"exclusively-penalized-q-learning-for-offline","title":"Exclusively Penalized Q-learning for Offline Reinforcement Learning","date":"2024-05-23","arxiv_id":"2405.14082","repositories_listed":0,"syntology":null},{"url":null,"slug":"privileged-sensing-scaffolds-reinforcement","title":"Privileged Sensing Scaffolds Reinforcement Learning","date":"2024-05-23","arxiv_id":"2405.14853","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-fine-tuning-text-1","title":"DLPO: Diffusion Model Loss-Guided Reinforcement Learning for Fine-Tuning Text-to-Speech Diffusion Models","date":"2024-05-23","arxiv_id":"2405.14632","repositories_listed":0,"syntology":null}],"record_sha256":"c1d9913fc1c05c280612beb5ddb854202f086a6c471394524cede88169789720","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}