{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/57","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":57,"pages_in_order":132,"rows_per_page":100,"rows":[5601,5700],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/56","next":"/task/reinforcement-learning/papers/58","papers":[{"url":null,"slug":"proxy-rlhf-decoupling-generation-and","title":"Proxy-RLHF: Decoupling Generation and Alignment in Large Language Model with Proxy","date":"2024-03-07","arxiv_id":"2403.04283","repositories_listed":0,"syntology":null},{"url":null,"slug":"teaching-large-language-models-to-reason-with","title":"Teaching Large Language Models to Reason with Reinforcement Learning","date":"2024-03-07","arxiv_id":"2403.04642","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-online-reinforcement-learning-is-causal","title":"Why Online Reinforcement Learning is Causal","date":"2024-03-07","arxiv_id":"2403.04221","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-applications-of-reinforcement","title":"A Survey on Applications of Reinforcement Learning in Spatial Resource Allocation","date":"2024-03-06","arxiv_id":"2403.03643","repositories_listed":0,"syntology":null},{"url":null,"slug":"reconciling-reality-through-simulation-a-real","title":"Reconciling Reality through Simulation: A Real-to-Sim-to-Real Approach for Robust Manipulation","date":"2024-03-06","arxiv_id":"2403.03949","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-zero-shot-reinforcement-learning-strategy","title":"A Zero-Shot Reinforcement Learning Strategy for Autonomous Guidewire Navigation","date":"2024-03-05","arxiv_id":"2403.02777","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-vehicle-decision-and-control","title":"Autonomous vehicle decision and control through reinforcement learning with traffic flow randomization","date":"2024-03-05","arxiv_id":"2403.02882","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-llm-safety-via-constrained-direct","title":"Enhancing LLM Safety via Constrained Direct Preference Optimization","date":"2024-03-04","arxiv_id":"2403.02475","repositories_listed":0,"syntology":null},{"url":null,"slug":"koopman-assisted-reinforcement-learning","title":"Koopman-Assisted Reinforcement Learning","date":"2024-03-04","arxiv_id":"2403.02290","repositories_listed":0,"syntology":null},{"url":null,"slug":"tsallis-entropy-regularization-for-linearly","title":"Tsallis Entropy Regularization for Linearly Solvable MDP and Linear Quadratic Regulator","date":"2024-03-04","arxiv_id":"2403.01805","repositories_listed":0,"syntology":null},{"url":null,"slug":"twisting-lids-off-with-two-hands","title":"Twisting Lids Off with Two Hands","date":"2024-03-04","arxiv_id":"2403.02338","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-myopic-exploration-through","title":"Sample Efficient Myopic Exploration Through Multitask Reinforcement Learning with Diverse Tasks","date":"2024-03-03","arxiv_id":"2403.01636","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-provable-log-density-policy-gradient","title":"Towards Provable Log Density Policy Gradient","date":"2024-03-03","arxiv_id":"2403.01605","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-mean-zero-disagreement-regularized","title":"Continuous Mean-Zero Disagreement-Regularized Imitation Learning (CMZ-DRIL)","date":"2024-03-02","arxiv_id":"2403.01059","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-role-of-information-structure-in","title":"On the Role of Information Structure in Reinforcement Learning for Partially-Observable Sequential Teams and Games","date":"2024-03-01","arxiv_id":"2403.00993","repositories_listed":0,"syntology":null},{"url":null,"slug":"scale-free-adversarial-reinforcement-learning","title":"Scale-free Adversarial Reinforcement Learning","date":"2024-03-01","arxiv_id":"2403.00930","repositories_listed":0,"syntology":null},{"url":null,"slug":"scale-invariant-gradient-aggregation-for","title":"Conflict-Averse Gradient Aggregation for Constrained Multi-Objective Reinforcement Learning","date":"2024-03-01","arxiv_id":"2403.00282","repositories_listed":0,"syntology":null},{"url":null,"slug":"selfi-autonomous-self-improvement-with","title":"SELFI: Autonomous Self-Improvement with Reinforcement Learning for Social Navigation","date":"2024-03-01","arxiv_id":"2403.00991","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-testing-environment-generation-for","title":"Adaptive Testing Environment Generation for Connected and Automated Vehicles with Dense Reinforcement Learning","date":"2024-02-29","arxiv_id":"2402.19275","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-a-convex","title":"Deep Reinforcement Learning: A Convex Optimization Approach","date":"2024-02-29","arxiv_id":"2402.19212","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-the-causes-of-plasticity-loss","title":"Disentangling the Causes of Plasticity Loss in Neural Networks","date":"2024-02-29","arxiv_id":"2402.18762","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-gpt-integrating-reinforcement-learning-and","title":"RL-GPT: Integrating Reinforcement Learning and Code-as-policy","date":"2024-02-29","arxiv_id":"2402.19299","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-centric-aware-uav-trajectory-planning","title":"Human-Centric Aware UAV Trajectory Planning in Search and Rescue Missions Employing Multi-Objective Reinforcement Learning with AHP and Similarity-Based Experience Replay","date":"2024-02-28","arxiv_id":"2402.18487","repositories_listed":0,"syntology":null},{"url":null,"slug":"implementing-online-reinforcement-learning-1","title":"Implementing Online Reinforcement Learning with Clustering Neural Networks","date":"2024-02-28","arxiv_id":"2402.18472","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-risk-sensitive-distributional","title":"Provable Risk-Sensitive Distributional Reinforcement Learning with General Function Approximation","date":"2024-02-28","arxiv_id":"2402.18159","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-partially-observable-risk","title":"Provably Efficient Partially Observable Risk-Sensitive Reinforcement Learning with Hindsight Observation","date":"2024-02-28","arxiv_id":"2402.18149","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-do-animals-need-shaping-a-theory-of-task","title":"Why Do Animals Need Shaping? A Theory of Task Composition and Curriculum Learning","date":"2024-02-28","arxiv_id":"2402.18361","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-investment-frontiers-industry-grade","title":"Advancing Investment Frontiers: Industry-grade Deep Reinforcement Learning for Portfolio Optimization","date":"2024-02-27","arxiv_id":"2403.07916","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for-16","title":"Multi-Agent Deep Reinforcement Learning for Distributed Satellite Routing","date":"2024-02-27","arxiv_id":"2402.17666","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-kubernetes-automated-scheduling","title":"Enhancing Kubernetes Automated Scheduling with Deep Learning and Reinforcement Techniques for Large-Scale Cloud Computing Optimization","date":"2024-02-26","arxiv_id":"2403.07905","repositories_listed":0,"syntology":null},{"url":null,"slug":"monitoring-fidelity-of-online-reinforcement","title":"Monitoring Fidelity of Online Reinforcement Learning Algorithms in Clinical Trials","date":"2024-02-26","arxiv_id":"2402.17003","repositories_listed":0,"syntology":null},{"url":null,"slug":"program-based-strategy-induction-for","title":"Program-Based Strategy Induction for Reinforcement Learning","date":"2024-02-26","arxiv_id":"2402.16668","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-fox-learning-breaking-tradition-in","title":"QF-tuner: Breaking Tradition in Reinforcement Learning","date":"2024-02-26","arxiv_id":"2402.16562","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-jazz-improvisation","title":"Reinforcement Learning Jazz Improvisation: When Music Meets Game Theory","date":"2024-02-25","arxiv_id":"2403.03224","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-off-policy-multi-step-td-learning","title":"Analysis of Off-Policy Multi-Step TD-Learning with Linear Function Approximation","date":"2024-02-24","arxiv_id":"2402.15781","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-optimized-reinforcement-learning-via","title":"Safety Optimized Reinforcement Learning via Multi-Objective Policy Optimization","date":"2024-02-23","arxiv_id":"2402.15197","repositories_listed":0,"syntology":null},{"url":null,"slug":"shapley-value-based-multi-agent-reinforcement","title":"Shapley Value Based Multi-Agent Reinforcement Learning: Theory, Method and Its Application to Energy Network","date":"2024-02-23","arxiv_id":"2402.15324","repositories_listed":0,"syntology":null},{"url":null,"slug":"trajectory-wise-iterative-reinforcement","title":"Trajectory-wise Iterative Reinforcement Learning Framework for Auto-bidding","date":"2024-02-23","arxiv_id":"2402.15102","repositories_listed":0,"syntology":null},{"url":null,"slug":"applying-reinforcement-learning-to-optimize","title":"Applying Reinforcement Learning to Optimize Traffic Light Cycles","date":"2024-02-22","arxiv_id":"2402.14886","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-control-of","title":"Model-Based Reinforcement Learning Control of Reaction-Diffusion Problems","date":"2024-02-22","arxiv_id":"2402.14446","repositories_listed":0,"syntology":null},{"url":null,"slug":"mr-arl-model-reference-adaptive-reinforcement","title":"MR-ARL: Model Reference Adaptive Reinforcement Learning for Robustly Stable On-Policy Data-Driven LQR","date":"2024-02-22","arxiv_id":"2402.14483","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-elastic-time","title":"Reinforcement Learning with Elastic Time Steps","date":"2024-02-22","arxiv_id":"2402.14961","repositories_listed":0,"syntology":null},{"url":null,"slug":"attackgnn-red-teaming-gnns-in-hardware","title":"AttackGNN: Red-Teaming GNNs in Hardware Security Using Reinforcement Learning","date":"2024-02-21","arxiv_id":"2402.13946","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-assisted-quantum-1","title":"Reinforcement learning-assisted quantum architecture search for variational quantum algorithms","date":"2024-02-21","arxiv_id":"2402.13754","repositories_listed":0,"syntology":null},{"url":null,"slug":"synthesis-of-hierarchical-controllers-based","title":"Synthesis of Hierarchical Controllers Based on Deep Reinforcement Learning Policies","date":"2024-02-21","arxiv_id":"2402.13785","repositories_listed":0,"syntology":null},{"url":null,"slug":"achieving-near-optimal-regret-for-bandit","title":"Uniform Last-Iterate Guarantee for Bandits and Reinforcement Learning","date":"2024-02-20","arxiv_id":"2402.12711","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-monocular-video-based-gait-analysis","title":"Advancing Monocular Video-Based Gait Analysis Using Motion Imitation with Physics-Based Simulation","date":"2024-02-20","arxiv_id":"2402.12676","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-hedging-with-market-impact","title":"Deep Hedging with Market Impact","date":"2024-02-20","arxiv_id":"2402.13326","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-reinforcement-learning-a","title":"Evolutionary Reinforcement Learning: A Systematic Review and Future Directions","date":"2024-02-20","arxiv_id":"2402.13296","repositories_listed":0,"syntology":null},{"url":null,"slug":"covrl-fuzzing-javascript-engines-with","title":"CovRL: Fuzzing JavaScript Engines with Coverage-Guided Reinforcement Learning for LLM-based Mutation","date":"2024-02-19","arxiv_id":"2402.12222","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-deep-reinforcement-learning-a-pruned","title":"In value-based deep reinforcement learning, a pruned network is a good network","date":"2024-02-19","arxiv_id":"2402.12479","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-optimal-execution","title":"Reinforcement Learning for Optimal Execution when Liquidity is Time-Varying","date":"2024-02-19","arxiv_id":"2402.12049","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-garment-recommendation-with-user","title":"Interactive Garment Recommendation with User in the Loop","date":"2024-02-18","arxiv_id":"2402.11627","repositories_listed":0,"syntology":null},{"url":null,"slug":"theoretical-foundations-for-programmatic","title":"Programmatic Reinforcement Learning: Navigating Gridworlds","date":"2024-02-18","arxiv_id":"2402.11650","repositories_listed":0,"syntology":null},{"url":null,"slug":"implementation-of-a-model-of-the-cortex-basal","title":"Implementation of a Model of the Cortex Basal Ganglia Loop","date":"2024-02-17","arxiv_id":"2402.13275","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-task-inverse-reinforcement-learning-for","title":"Multi Task Inverse Reinforcement Learning for Common Sense Reward","date":"2024-02-17","arxiv_id":"2402.11367","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-to-maximise-wind","title":"Reinforcement learning to maximise wind turbine energy generation","date":"2024-02-17","arxiv_id":"2402.11384","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-constrained-clustering-with","title":"Large Scale Constrained Clustering With Reinforcement Learning","date":"2024-02-15","arxiv_id":"2402.10177","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-solving-stochastic-1","title":"Reinforcement Learning for Solving Stochastic Vehicle Routing Problem with Time Windows","date":"2024-02-15","arxiv_id":"2402.09765","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-poisoning-attack-against-offline","title":"Universal Black-Box Reward Poisoning Attack against Offline Reinforcement Learning","date":"2024-02-15","arxiv_id":"2402.09695","repositories_listed":0,"syntology":null},{"url":null,"slug":"smart-information-exchange-for-unsupervised","title":"Smart Information Exchange for Unsupervised Federated Learning via Reinforcement Learning","date":"2024-02-15","arxiv_id":"2402.09629","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-interpretable-policies-in-hindsight","title":"Learning Interpretable Policies in Hindsight-Observable POMDPs through Partially Supervised Reinforcement Learning","date":"2024-02-14","arxiv_id":"2402.09290","repositories_listed":0,"syntology":null},{"url":null,"slug":"ll-gabr-energy-efficient-live-video-streaming","title":"LL-GABR: Energy Efficient Live Video Streaming Using Reinforcement Learning","date":"2024-02-14","arxiv_id":"2402.09392","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-exploration-in-reinforcement","title":"How does Your RL Agent Explore? An Optimal Transport Analysis of Occupancy Measure Trajectories","date":"2024-02-14","arxiv_id":"2402.09113","repositories_listed":0,"syntology":null},{"url":null,"slug":"pmgda-a-preference-based-multiple-gradient","title":"PMGDA: A Preference-based Multiple Gradient Descent Algorithm","date":"2024-02-14","arxiv_id":"2402.09492","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-from-human-feedback","title":"Reinforcement Learning from Human Feedback with Active Queries","date":"2024-02-14","arxiv_id":"2402.09401","repositories_listed":0,"syntology":null},{"url":null,"slug":"steady-state-error-compensation-for","title":"Steady-State Error Compensation for Reinforcement Learning with Quadratic Rewards","date":"2024-02-14","arxiv_id":"2402.09075","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-model-based-reinforcement","title":"Towards Robust Model-Based Reinforcement Learning Against Adversarial Corruption","date":"2024-02-14","arxiv_id":"2402.08991","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-transient-stability","title":"Uncertainty-Aware Transient Stability-Constrained Preventive Redispatch: A Distributional Reinforcement Learning Approach","date":"2024-02-14","arxiv_id":"2402.09263","repositories_listed":0,"syntology":null},{"url":null,"slug":"ir-aware-eco-timing-optimization-using","title":"IR-Aware ECO Timing Optimization Using Reinforcement Learning","date":"2024-02-12","arxiv_id":"2402.07781","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-as-agents-in-two-player","title":"Large Language Models as Agents in Two-Player Games","date":"2024-02-12","arxiv_id":"2402.08078","repositories_listed":0,"syntology":null},{"url":null,"slug":"maidcrl-semi-centralized-multi-agent","title":"MAIDCRL: Semi-centralized Multi-Agent Influence Dense-CNN Reinforcement Learning","date":"2024-02-12","arxiv_id":"2402.07890","repositories_listed":0,"syntology":null},{"url":null,"slug":"measurement-scheduling-for-icu-patients-with","title":"Measurement Scheduling for ICU Patients with Offline Reinforcement Learning","date":"2024-02-12","arxiv_id":"2402.07344","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-minimax-optimal-distributional","title":"Near-Minimax-Optimal Distributional Reinforcement Learning with a Generative Model","date":"2024-02-12","arxiv_id":"2402.07598","repositories_listed":0,"syntology":null},{"url":null,"slug":"future-prediction-can-be-a-strong-evidence-of","title":"Future Prediction Can be a Strong Evidence of Good History Representation in Partially Observable Environments","date":"2024-02-11","arxiv_id":"2402.07102","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-language-reinforcement-learning","title":"Natural Language Reinforcement Learning","date":"2024-02-11","arxiv_id":"2402.07157","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-generalized-inverse-reinforcement","title":"Towards Generalized Inverse Reinforcement Learning","date":"2024-02-11","arxiv_id":"2402.07246","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-large-language-models-to-automate-and","title":"Using Large Language Models to Automate and Expedite Reinforcement Learning with Reward Machine","date":"2024-02-11","arxiv_id":"2402.07069","repositories_listed":0,"syntology":null},{"url":null,"slug":"principled-penalty-based-methods-for-bilevel","title":"Principled Penalty-based Methods for Bilevel Reinforcement Learning and RLHF","date":"2024-02-10","arxiv_id":"2402.06886","repositories_listed":0,"syntology":null},{"url":null,"slug":"corruption-robust-offline-reinforcement-1","title":"Corruption Robust Offline Reinforcement Learning with Human Feedback","date":"2024-02-09","arxiv_id":"2402.06734","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-transformers-are-efficient-meta","title":"Hierarchical Transformers are Efficient Meta-Reinforcement Learners","date":"2024-02-09","arxiv_id":"2402.06402","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-precision-geosteering-via-reinforcement","title":"High-Precision Geosteering via Reinforcement Learning and Particle Filters","date":"2024-02-09","arxiv_id":"2402.06377","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-function-interference-and-greedy-action","title":"Value function interference and greedy action selection in value-based multi-objective reinforcement learning","date":"2024-02-09","arxiv_id":"2402.06266","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentially-private-model-based-offline","title":"Differentially Private Deep Model-Based Reinforcement Learning","date":"2024-02-08","arxiv_id":"2402.05525","repositories_listed":0,"syntology":null},{"url":null,"slug":"edge-caching-based-on-deep-reinforcement","title":"Attention-Enhanced Prioritized Proximal Policy Optimization for Adaptive Edge Caching","date":"2024-02-08","arxiv_id":"2402.14576","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-offline-reinforcement-learning-1","title":"Federated Offline Reinforcement Learning: Collaborative Single-Policy Coverage Suffices","date":"2024-02-08","arxiv_id":"2402.05876","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-actor-critic-reinforcement-learning","title":"Offline Actor-Critic Reinforcement Learning Scales to Large Models","date":"2024-02-08","arxiv_id":"2402.05546","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-artificial-intelligence-for-digital","title":"Scaling Artificial Intelligence for Digital Wargaming in Support of Decision-Making","date":"2024-02-08","arxiv_id":"2402.06075","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-computational-approach-to-visual-ecology","title":"A computational approach to visual ecology with deep reinforcement learning","date":"2024-02-07","arxiv_id":"2402.05266","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-adversarial-inputs-in-deep","title":"Analyzing Adversarial Inputs in Deep Reinforcement Learning","date":"2024-02-07","arxiv_id":"2402.05284","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-as-reward-empowering-reinforcement","title":"Code as Reward: Empowering Reinforcement Learning with VLMs","date":"2024-02-07","arxiv_id":"2402.04764","repositories_listed":0,"syntology":null},{"url":"/paper/compound-returns-reduce-variance-in","slug":"compound-returns-reduce-variance-in","title":"Averaging $n$-step Returns Reduces Variance in Reinforcement Learning","date":"2024-02-06","arxiv_id":"2402.03903","repositories_listed":0,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/compound-returns-reduce-variance-in#ran","syntology_url":"https://syntology.ai/paper/2402.03903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03903"}},"official":null}},{"url":null,"slug":"no-regret-reinforcement-learning-in-smooth","title":"No-Regret Reinforcement Learning in Smooth MDPs","date":"2024-02-06","arxiv_id":"2402.03792","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-from-bagged-reward-a","title":"Reinforcement Learning from Bagged Reward","date":"2024-02-06","arxiv_id":"2402.03771","repositories_listed":0,"syntology":null},{"url":null,"slug":"transductive-reward-inference-on-graph","title":"Transductive Reward Inference on Graph","date":"2024-02-06","arxiv_id":"2402.03661","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multi-step-loss-function-for-robust","title":"A Multi-step Loss Function for Robust Learning of the Dynamics in Model-based Reinforcement Learning","date":"2024-02-05","arxiv_id":"2402.03146","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-for-dynamic","title":"A Reinforcement Learning Approach for Dynamic Rebalancing in Bike-Sharing System","date":"2024-02-05","arxiv_id":"2402.03589","repositories_listed":0,"syntology":null},{"url":null,"slug":"abstracted-trajectory-visualization-for","title":"Abstracted Trajectory Visualization for Explainability in Reinforcement Learning","date":"2024-02-05","arxiv_id":"2402.07928","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-the-impact-of-distribution-shift-on","title":"Assessing the Impact of Distribution Shift on Reinforcement Learning Performance","date":"2024-02-05","arxiv_id":"2402.03590","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-reinforcement-learning-for-quantum","title":"Curriculum reinforcement learning for quantum architecture search under hardware errors","date":"2024-02-05","arxiv_id":"2402.03500","repositories_listed":0,"syntology":null}],"record_sha256":"4c967e8fcdfa3200d9b7ade3c3bbb53c7e7fd18cf6f40a5da994154020784978","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}