{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/61","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":61,"pages_in_order":152,"rows_per_page":100,"rows":[6001,6100],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/60","next":"/task/reinforcement-learning-1/papers/62","papers":[{"url":null,"slug":"pessimism-meets-risk-risk-sensitive-offline","title":"Pessimism Meets Risk: Risk-Sensitive Offline Reinforcement Learning","date":"2024-07-10","arxiv_id":"2407.07631","repositories_listed":0,"syntology":null},{"url":null,"slug":"token-mol-1-0-tokenized-drug-design-with","title":"Token-Mol 1.0: Tokenized drug design with large language model","date":"2024-07-10","arxiv_id":"2407.07930","repositories_listed":0,"syntology":null},{"url":null,"slug":"intercepting-unauthorized-aerial-robots-in","title":"Intercepting Unauthorized Aerial Robots in Controlled Airspace Using Reinforcement Learning","date":"2024-07-09","arxiv_id":"2407.06909","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-open-source-multi-agent-deep-reinforcement","title":"An open source Multi-Agent Deep Reinforcement Learning Routing Simulator for satellite networks","date":"2024-07-08","arxiv_id":"2407.11047","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-based-5","title":"Multi-agent Reinforcement Learning-based Network Intrusion Detection System","date":"2024-07-08","arxiv_id":"2407.05766","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-bellman-equations-for-continuous-time","title":"On Bellman equations for continuous-time policy evaluation I: discretization and approximation","date":"2024-07-08","arxiv_id":"2407.05966","repositories_listed":0,"syntology":null},{"url":null,"slug":"periodic-agent-state-based-q-learning-for","title":"Periodic agent-state based Q-learning for POMDPs","date":"2024-07-08","arxiv_id":"2407.06121","repositories_listed":0,"syntology":null},{"url":null,"slug":"fosp-fine-tuning-offline-safe-policy-through","title":"FOSP: Fine-tuning Offline Safe Policy through World Models","date":"2024-07-06","arxiv_id":"2407.04942","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-off-policy-actor-critic","title":"Multi-agent Off-policy Actor-Critic Reinforcement Learning for Partially Observable Environments","date":"2024-07-06","arxiv_id":"2407.04974","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoverse-an-evolvable-game-langugage-for","title":"Autoverse: An Evolvable Game Language for Learning Robust Embodied Agents","date":"2024-07-05","arxiv_id":"2407.04221","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-decision-transformer-tackling-data","title":"Robust Decision Transformer: Tackling Data Corruption in Offline RL via Sequence Modeling","date":"2024-07-05","arxiv_id":"2407.04285","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-petri-nets-as-an-integrated-constraint","title":"Using Petri Nets as an Integrated Constraint Mechanism for Reinforcement Learning Tasks","date":"2024-07-05","arxiv_id":"2407.04481","repositories_listed":0,"syntology":null},{"url":null,"slug":"warm-up-free-policy-optimization-improved","title":"Warm-up Free Policy Optimization: Improved Regret in Linear Markov Decision Processes","date":"2024-07-03","arxiv_id":"2407.03065","repositories_listed":0,"syntology":null},{"url":null,"slug":"pwm-policy-learning-with-large-world-models","title":"PWM: Policy Learning with Multi-Task World Models","date":"2024-07-02","arxiv_id":"2407.02466","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-driven-data-intensive","title":"Reinforcement Learning-driven Data-intensive Workflow Scheduling for Volunteer Edge-Cloud","date":"2024-07-01","arxiv_id":"2407.01428","repositories_listed":0,"syntology":null},{"url":null,"slug":"to-switch-or-not-to-switch-balanced-policy","title":"To Switch or Not to Switch? Balanced Policy Switching in Offline Reinforcement Learning","date":"2024-07-01","arxiv_id":"2407.01837","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarks-for-reinforcement-learning-with","title":"Benchmarks for Reinforcement Learning with Biased Offline Data and Imperfect Simulators","date":"2024-06-30","arxiv_id":"2407.00806","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-for-power-system","title":"Safe Reinforcement Learning for Power System Control: A Review","date":"2024-06-30","arxiv_id":"2407.00681","repositories_listed":0,"syntology":null},{"url":null,"slug":"tackling-long-horizon-tasks-with-model-based","title":"Model-based Offline Reinforcement Learning with Lower Expectile Q-Learning","date":"2024-06-30","arxiv_id":"2407.00699","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-safe-reinforcement-learning-1","title":"A Review of Safe Reinforcement Learning Methods for Modern Power Systems","date":"2024-06-29","arxiv_id":"2407.00304","repositories_listed":0,"syntology":null},{"url":null,"slug":"digital-twin-assisted-data-driven","title":"Digital Twin-Assisted Data-Driven Optimization for Reliable Edge Caching in Wireless Networks","date":"2024-06-29","arxiv_id":"2407.00286","repositories_listed":0,"syntology":null},{"url":null,"slug":"medical-knowledge-integration-into","title":"Medical Knowledge Integration into Reinforcement Learning Algorithms for Dynamic Treatment Regimes","date":"2024-06-29","arxiv_id":"2407.00364","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-human-preferences-exploring","title":"Beyond Human Preferences: Exploring Reinforcement Learning Trajectory Evaluation and Improvement through LLMs","date":"2024-06-28","arxiv_id":"2406.19644","repositories_listed":0,"syntology":null},{"url":null,"slug":"decision-transformer-for-irs-assisted-systems","title":"Decision Transformer for IRS-Assisted Systems with Diffusion-Driven Generative Channels","date":"2024-06-28","arxiv_id":"2406.19769","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-cyber-defense-in-dynamic-active","title":"Optimizing Cyber Defense in Dynamic Active Directories through Reinforcement Learning","date":"2024-06-28","arxiv_id":"2406.19596","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-efficient-design","title":"Reinforcement Learning for Efficient Design and Control Co-optimisation of Energy Systems","date":"2024-06-28","arxiv_id":"2406.19825","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-policy-gradient-aligning-llms-on","title":"Contrastive Policy Gradient: Aligning LLMs on sequence-level scores in a supervised-friendly fashion","date":"2024-06-27","arxiv_id":"2406.19185","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-gradient-search-control-a-method-for","title":"Meta-Gradient Search Control: A Method for Improving the Efficiency of Dyna-style Planning","date":"2024-06-27","arxiv_id":"2406.19561","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-semantic-traffic-control-in-avs","title":"Decentralized Semantic Traffic Control in AVs Using RL and DQN for Dynamic Roadblocks","date":"2024-06-26","arxiv_id":"2406.18741","repositories_listed":0,"syntology":null},{"url":null,"slug":"preference-elicitation-for-offline","title":"Preference Elicitation for Offline Reinforcement Learning","date":"2024-06-26","arxiv_id":"2406.18450","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-intrinsically","title":"Reinforcement Learning with Intrinsically Motivated Feedback Graph for Lost-sales Inventory Control","date":"2024-06-26","arxiv_id":"2406.18351","repositories_listed":0,"syntology":null},{"url":null,"slug":"extract-efficient-policy-learning-by","title":"EXTRACT: Efficient Policy Learning by Extracting Transferable Robot Skills from Offline Data","date":"2024-06-25","arxiv_id":"2406.17768","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-object-interaction-from-human-level","title":"Human-Object Interaction from Human-Level Instructions","date":"2024-06-25","arxiv_id":"2406.17840","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-reinforcement-learning-in-red","title":"Leveraging Reinforcement Learning in Red Teaming for Advanced Ransomware Attack Simulations","date":"2024-06-25","arxiv_id":"2406.17576","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-preserving-reinforcement-learning-for","title":"Privacy Preserving Reinforcement Learning for Population Processes","date":"2024-06-25","arxiv_id":"2406.17649","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-state-action-reward-state-action","title":"The State-Action-Reward-State-Action Algorithm in Spatial Prisoner's Dilemma Game","date":"2024-06-25","arxiv_id":"2406.17326","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-rl-based-data-transmission","title":"Decentralized RL-Based Data Transmission Scheme for Energy Efficient Harvesting","date":"2024-06-24","arxiv_id":"2406.16624","repositories_listed":0,"syntology":null},{"url":null,"slug":"ocalm-object-centric-assessment-with-language","title":"OCALM: Object-Centric Assessment with Language Models","date":"2024-06-24","arxiv_id":"2406.16748","repositories_listed":0,"syntology":null},{"url":null,"slug":"tolerance-of-reinforcement-learning","title":"Tolerance of Reinforcement Learning Controllers against Deviations in Cyber Physical Systems","date":"2024-06-24","arxiv_id":"2406.17066","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-spectral-representation-for","title":"Diffusion Spectral Representation for Reinforcement Learning","date":"2024-06-23","arxiv_id":"2406.16121","repositories_listed":0,"syntology":null},{"url":null,"slug":"multistep-criticality-search-and-power","title":"Multistep Criticality Search and Power Shaping in Microreactors with Reinforcement Learning","date":"2024-06-22","arxiv_id":"2406.15931","repositories_listed":0,"syntology":null},{"url":null,"slug":"kalmamba-towards-efficient-probabilistic","title":"KalMamba: Towards Efficient Probabilistic State Space Models for RL under Uncertainty","date":"2024-06-21","arxiv_id":"2406.15131","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-problem-order-optimal-regret-bounds-for","title":"Open Problem: Order Optimal Regret Bounds for Kernel-Based Reinforcement Learning","date":"2024-06-21","arxiv_id":"2406.15250","repositories_listed":0,"syntology":null},{"url":null,"slug":"equivariant-offline-reinforcement-learning","title":"Equivariant Offline Reinforcement Learning","date":"2024-06-20","arxiv_id":"2406.13961","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-novelty-of-top-k-recommendations","title":"Optimizing Novelty of Top-k Recommendations using Large Language Models and Reinforcement Learning","date":"2024-06-20","arxiv_id":"2406.14169","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-optimization-for-tail-based-control","title":"Resource Optimization for Tail-Based Control in Wireless Networked Control Systems","date":"2024-06-20","arxiv_id":"2406.14301","repositories_listed":0,"syntology":null},{"url":null,"slug":"revealing-the-learning-process-in","title":"Revealing the learning process in reinforcement learning agents through attention-oriented metrics","date":"2024-06-20","arxiv_id":"2406.14324","repositories_listed":0,"syntology":null},{"url":null,"slug":"rewarding-what-matters-step-by-step","title":"Rewarding What Matters: Step-by-Step Reinforcement Learning for Task-Oriented Dialogue","date":"2024-06-20","arxiv_id":"2406.14457","repositories_listed":0,"syntology":null},{"url":null,"slug":"urban-focused-multi-task-offline","title":"Urban-Focused Multi-Task Offline Reinforcement Learning with Contrastive Data Sharing","date":"2024-06-20","arxiv_id":"2406.14054","repositories_listed":0,"syntology":null},{"url":null,"slug":"learned-graph-rewriting-with-equality","title":"Learned Graph Rewriting with Equality Saturation: A New Paradigm in Relational Query Rewrite and Beyond","date":"2024-06-19","arxiv_id":"2407.12794","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-wireless-discontinuous-reception","title":"Optimizing Wireless Discontinuous Reception via MAC Signaling Learning","date":"2024-06-19","arxiv_id":"2406.13834","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-safe-reinforcement-learning-enabled","title":"Adaptive Safe Reinforcement Learning-Enabled Optimization of Battery Fast-Charging Protocols","date":"2024-06-18","arxiv_id":"2406.12309","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-navigation-of-catheters-and","title":"Autonomous navigation of catheters and guidewires in mechanical thrombectomy using inverse reinforcement learning","date":"2024-06-18","arxiv_id":"2406.12499","repositories_listed":0,"syntology":null},{"url":null,"slug":"order-optimal-instance-dependent-bounds-for","title":"Order-Optimal Instance-Dependent Bounds for Offline Reinforcement Learning with Preference Feedback","date":"2024-06-18","arxiv_id":"2406.12205","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-compiling-with-reinforcement-learning","title":"Quantum Compiling with Reinforcement Learning on a Superconducting Processor","date":"2024-06-18","arxiv_id":"2406.12195","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-imitative-multi-token","title":"Physics-informed Imitative Reinforcement Learning for Real-world Driving","date":"2024-06-18","arxiv_id":"2407.02508","repositories_listed":0,"syntology":null},{"url":"/paper/adding-conditional-control-to-diffusion","slug":"adding-conditional-control-to-diffusion","title":"Adding Conditional Control to Diffusion Models with Reinforcement Learning","date":"2024-06-17","arxiv_id":"2406.12120","repositories_listed":0,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":9,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/adding-conditional-control-to-diffusion#ran","syntology_url":"https://syntology.ai/paper/2406.12120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12120"}},"official":null}},{"url":null,"slug":"constrained-reinforcement-learning-with-2","title":"Constrained Reinforcement Learning with Average Reward Objective: Model-Based and Model-Free Algorithms","date":"2024-06-17","arxiv_id":"2406.11481","repositories_listed":0,"syntology":null},{"url":null,"slug":"constructing-ancestral-recombination-graphs","title":"Constructing Ancestral Recombination Graphs through Reinforcement Learning","date":"2024-06-17","arxiv_id":"2406.12022","repositories_listed":0,"syntology":null},{"url":null,"slug":"linear-bellman-completeness-suffices-for","title":"Linear Bellman Completeness Suffices for Efficient Online Reinforcement Learning with Few Actions","date":"2024-06-17","arxiv_id":"2406.11640","repositories_listed":0,"syntology":null},{"url":null,"slug":"run-time-assured-reinforcement-learning-for","title":"Run Time Assured Reinforcement Learning for Six Degree-of-Freedom Spacecraft Inspection","date":"2024-06-17","arxiv_id":"2406.11795","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-of-interacting-particle-systems-for","title":"Design of Interacting Particle Systems for Fast Linear Quadratic RL","date":"2024-06-16","arxiv_id":"2406.11057","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-and-evolving-reward-functions-for","title":"Generating and Evolving Reward Functions for Highway Driving with Large Language Models","date":"2024-06-15","arxiv_id":"2406.10540","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-time-analysis-of-simultaneous-double-q","title":"Finite-Time Analysis of Simultaneous Double Q-learning","date":"2024-06-14","arxiv_id":"2406.09946","repositories_listed":0,"syntology":null},{"url":null,"slug":"roar-reinforcing-original-to-augmented-data","title":"ROAR: Reinforcing Original to Augmented Data Ratio Dynamics for Wav2Vec2.0 Based ASR","date":"2024-06-14","arxiv_id":"2406.09999","repositories_listed":0,"syntology":null},{"url":null,"slug":"unlock-the-correlation-between-supervised","title":"Unlock the Correlation between Supervised Fine-Tuning and Reinforcement Learning in Training Code Large Language Models","date":"2024-06-14","arxiv_id":"2406.10305","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-actor-critic-based-optimal","title":"Adaptive Actor-Critic Based Optimal Regulation for Drift-Free Uncertain Nonlinear Systems","date":"2024-06-13","arxiv_id":"2406.09097","repositories_listed":0,"syntology":null},{"url":null,"slug":"cimrl-combining-imitiation-and-reinforcement","title":"CIMRL: Combining IMitation and Reinforcement Learning for Safe Autonomous Driving","date":"2024-06-13","arxiv_id":"2406.08878","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-modeling-and-supervisory-control","title":"Data-driven modeling and supervisory control system optimization for plug-in hybrid electric vehicles","date":"2024-06-13","arxiv_id":"2406.09082","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffpogan-diffusion-policies-with-generative","title":"DiffPoGAN: Diffusion Policies with Generative Adversarial Networks for Offline Reinforcement Learning","date":"2024-06-13","arxiv_id":"2406.09089","repositories_listed":0,"syntology":null},{"url":null,"slug":"e-cop-episodic-constrained-optimization-of","title":"e-COP : Episodic Constrained Optimization of Policies","date":"2024-06-13","arxiv_id":"2406.09563","repositories_listed":0,"syntology":null},{"url":null,"slug":"semopo-learning-high-quality-model-and-policy","title":"SeMOPO: Learning High-quality Model and Policy from Low-quality Offline Visual Datasets","date":"2024-06-13","arxiv_id":"2406.09486","repositories_listed":0,"syntology":null},{"url":null,"slug":"rile-reinforced-imitation-learning","title":"RILe: Reinforced Imitation Learning","date":"2024-06-12","arxiv_id":"2406.08472","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-value-iteration-networks-to-5000","title":"Scaling Value Iteration Networks to 5000 Layers for Extreme Long-Term Planning","date":"2024-06-12","arxiv_id":"2406.08404","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-enhanced-reinforcement-learning-based","title":"Toward Enhanced Reinforcement Learning-Based Resource Management via Digital Twin: Opportunities, Applications, and Challenges","date":"2024-06-12","arxiv_id":"2406.07857","repositories_listed":0,"syntology":null},{"url":null,"slug":"charme-a-chain-based-reinforcement-learning","title":"CHARME: A chain-based reinforcement learning approach for the minor embedding problem","date":"2024-06-11","arxiv_id":"2406.07124","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-gene-selection-in-single-cell","title":"Enhanced Gene Selection in Single-Cell Genomics: Pre-Filtering Synergy and Reinforced Optimization","date":"2024-06-11","arxiv_id":"2406.07418","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-reinforcement-learning-from-offline","title":"Hybrid Reinforcement Learning from Offline Observation Alone","date":"2024-06-11","arxiv_id":"2406.07253","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-domain-knowledge-for-handling","title":"Integrating Domain Knowledge for handling Limited Data in Offline RL","date":"2024-06-11","arxiv_id":"2406.07041","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-demonstration-and-preference-learning","title":"Learning Reward and Policy Jointly from Demonstration and Preference Improves Alignment","date":"2024-06-11","arxiv_id":"2406.06874","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-complexity-reduction-via-policy","title":"Sample Complexity Reduction via Policy Difference Estimation in Tabular Reinforcement Learning","date":"2024-06-11","arxiv_id":"2406.06856","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-multiple-solutions-from-a-single","title":"Discovering Multiple Solutions from a Single Task in Offline Reinforcement Learning","date":"2024-06-10","arxiv_id":"2406.05993","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-based-reinforcement-learning-for","title":"Diffusion-based Reinforcement Learning for Dynamic UAV-assisted Vehicle Twins Migration in Vehicular Metaverses","date":"2024-06-08","arxiv_id":"2406.05422","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhanced-flight-envelope-protection-a-novel","title":"Enhanced Flight Envelope Protection: A Novel Reinforcement Learning Approach","date":"2024-06-08","arxiv_id":"2406.05586","repositories_listed":0,"syntology":null},{"url":"/paper/optimizing-automatic-differentiation-with","slug":"optimizing-automatic-differentiation-with","title":"Optimizing Automatic Differentiation with Deep Reinforcement Learning","date":"2024-06-07","arxiv_id":"2406.05027","repositories_listed":0,"syntology":{"n":29,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":17,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 17 unverified","sample_list":"/paper/optimizing-automatic-differentiation-with#ran","syntology_url":"https://syntology.ai/paper/2406.05027","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.05027"}},"official":null}},{"url":null,"slug":"primitive-agentic-first-order-optimization","title":"Primitive Agentic First-Order Optimization","date":"2024-06-07","arxiv_id":"2406.04841","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-to-real-transfer-of-deep-reinforcement","title":"Sim-to-Real Transfer of Deep Reinforcement Learning Agents for Online Coverage Path Planning","date":"2024-06-07","arxiv_id":"2406.04920","repositories_listed":0,"syntology":null},{"url":null,"slug":"atradiff-accelerating-online-reinforcement","title":"ATraDiff: Accelerating Online Reinforcement Learning with Imaginary Trajectories","date":"2024-06-06","arxiv_id":"2406.04323","repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrapping-expectiles-in-reinforcement","title":"Bootstrapping Expectiles in Reinforcement Learning","date":"2024-06-06","arxiv_id":"2406.04081","repositories_listed":0,"syntology":null},{"url":null,"slug":"breeding-programs-optimization-with","title":"Breeding Programs Optimization with Reinforcement Learning","date":"2024-06-06","arxiv_id":"2406.03932","repositories_listed":0,"syntology":null},{"url":null,"slug":"excluding-the-irrelevant-focusing","title":"Excluding the Irrelevant: Focusing Reinforcement Learning through Continuous Action Masking","date":"2024-06-06","arxiv_id":"2406.03704","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-autonomous-driving-for-safety-a","title":"Optimizing Autonomous Driving for Safety: A Human-Centric Approach with LLM-Enhanced RLHF","date":"2024-06-06","arxiv_id":"2406.04481","repositories_listed":0,"syntology":null},{"url":null,"slug":"proofread-fixes-all-errors-with-one-tap","title":"Proofread: Fixes All Errors with One Tap","date":"2024-06-06","arxiv_id":"2406.04523","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-play-with-adversarial-critic-provable","title":"Self-Play with Adversarial Critic: Provable and Scalable Offline Alignment for Language Models","date":"2024-06-06","arxiv_id":"2406.04274","repositories_listed":0,"syntology":null},{"url":null,"slug":"deer-a-delay-resilient-framework-for","title":"DEER: A Delay-Resilient Framework for Reinforcement Learning with Variable Delays","date":"2024-06-05","arxiv_id":"2406.03102","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-tarzan-to-tolkien-controlling-the","title":"From Tarzan to Tolkien: Controlling the Language Proficiency Level of LLMs for Content Generation","date":"2024-06-05","arxiv_id":"2406.03030","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-based-visual-alignment-for-zero-shot","title":"Prompt-based Visual Alignment for Zero-shot Policy Transfer","date":"2024-06-05","arxiv_id":"2406.03250","repositories_listed":0,"syntology":null},{"url":"/paper/scaling-laws-for-reward-model-1","slug":"scaling-laws-for-reward-model-1","title":"Scaling Laws for Reward Model Overoptimization in Direct Alignment Algorithms","date":"2024-06-05","arxiv_id":"2406.02900","repositories_listed":0,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":8,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":10,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scaling-laws-for-reward-model-1#ran","syntology_url":"https://syntology.ai/paper/2406.02900","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.02900"}},"official":null}},{"url":null,"slug":"udql-bridging-the-gap-between-mse-loss-and","title":"UDQL: Bridging The Gap between MSE Loss and The Optimal Value Function in Offline Reinforcement Learning","date":"2024-06-05","arxiv_id":"2406.03324","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unifying-framework-for-action-conditional","title":"A Unifying Framework for Action-Conditional Self-Predictive Reinforcement Learning","date":"2024-06-04","arxiv_id":"2406.02035","repositories_listed":0,"syntology":null}],"record_sha256":"a5bb8925f559333446535386f03bcc751c43e974450981ba38ce222c8108b2d1","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}