{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/56","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":56,"pages_in_order":132,"rows_per_page":100,"rows":[5501,5600],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/55","next":"/task/reinforcement-learning/papers/57","papers":[{"url":null,"slug":"higher-replay-ratio-empowers-sample-efficient","title":"Higher Replay Ratio Empowers Sample-Efficient Multi-Agent Reinforcement Learning","date":"2024-04-15","arxiv_id":"2404.09715","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effects-of-fine-tuning-language-models","title":"On the Effects of Fine-tuning Language Models for Text-Based Reinforcement Learning","date":"2024-04-15","arxiv_id":"2404.10174","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledgeable-agents-by-offline-reinforcement","title":"Knowledgeable Agents by Offline Reinforcement Learning from Large Language Model Rollouts","date":"2024-04-14","arxiv_id":"2404.09248","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-offline-quantum-reinforcement","title":"Model-based Offline Quantum Reinforcement Learning","date":"2024-04-14","arxiv_id":"2404.10017","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-learning-for-control-oriented","title":"Active Learning for Control-Oriented Identification of Nonlinear Systems","date":"2024-04-13","arxiv_id":"2404.09030","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-on-the-constraint","title":"Safe Reinforcement Learning on the Constraint Manifold: Theory and Applications","date":"2024-04-13","arxiv_id":"2404.09080","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-forest-fire-prevention-deep","title":"Advancing Forest Fire Prevention: Deep Reinforcement Learning for Effective Firebreak Placement","date":"2024-04-12","arxiv_id":"2404.08523","repositories_listed":0,"syntology":null},{"url":null,"slug":"agile-and-versatile-bipedal-robot-tracking","title":"Agile and versatile bipedal robot tracking control through reinforcement learning","date":"2024-04-12","arxiv_id":"2404.08246","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlemmo-evolutionary-multimodal-optimization","title":"RLEMMO: Evolutionary Multimodal Optimization Assisted By Deep Reinforcement Learning","date":"2024-04-12","arxiv_id":"2404.08242","repositories_listed":0,"syntology":null},{"url":null,"slug":"sir-rl-reinforcement-learning-for-optimized","title":"SIR-RL: Reinforcement Learning for Optimized Policy Control during Epidemiological Outbreaks in Emerging Market and Developing Economies","date":"2024-04-12","arxiv_id":"2404.08423","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentially-private-reinforcement-learning-1","title":"Differentially Private Reinforcement Learning with Self-Play","date":"2024-04-11","arxiv_id":"2404.07559","repositories_listed":0,"syntology":null},{"url":null,"slug":"fpga-divide-and-conquer-placement-using-deep","title":"FPGA Divide-and-Conquer Placement using Deep Reinforcement Learning","date":"2024-04-11","arxiv_id":"2404.13061","repositories_listed":0,"syntology":null},{"url":null,"slug":"r2-indicator-and-deep-reinforcement-learning","title":"R2 Indicator and Deep Reinforcement Learning Enhanced Adaptive Multi-Objective Evolutionary Algorithm","date":"2024-04-11","arxiv_id":"2404.08161","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-enabled-collaborative-beamforming-via","title":"UAV-enabled Collaborative Beamforming via Multi-Agent Deep Reinforcement Learning","date":"2024-04-11","arxiv_id":"2404.07453","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-reinforcement-learning-for-media","title":"Structured Reinforcement Learning for Media Streaming at the Wireless Edge","date":"2024-04-10","arxiv_id":"2404.07315","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-approach","title":"Deep Reinforcement Learning-Based Approach for a Single Vehicle Persistent Surveillance Problem with Fuel Constraints","date":"2024-04-09","arxiv_id":"2404.06423","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-multi-task-reinforcement-learning-1","title":"Efficient Multi-Task Reinforcement Learning via Task-Specific Action Correction","date":"2024-04-09","arxiv_id":"2404.05950","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-pre-trained-transformer-for-2","title":"Generative Pre-Trained Transformer for Symbolic Regression Base In-Context Reinforcement Learning","date":"2024-04-09","arxiv_id":"2404.06330","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-reinforcement-learning-for","title":"Graph Reinforcement Learning for Combinatorial Optimization: A Survey and Unifying Perspective","date":"2024-04-09","arxiv_id":"2404.06492","repositories_listed":0,"syntology":null},{"url":null,"slug":"chiplet-placement-order-exploration-based-on","title":"Chiplet Placement Order Exploration Based on Learning to Rank with Graph Representation","date":"2024-04-07","arxiv_id":"2404.04943","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-control-for","title":"Deep Reinforcement Learning Control for Disturbance Rejection in a Nonlinear Dynamic System with Parametric Uncertainty","date":"2024-04-06","arxiv_id":"2404.04699","repositories_listed":0,"syntology":null},{"url":null,"slug":"structurally-flexible-neural-networks","title":"Structurally Flexible Neural Networks: Evolving the Building Blocks for General Agents","date":"2024-04-06","arxiv_id":"2404.15193","repositories_listed":0,"syntology":null},{"url":null,"slug":"demonstration-guided-multi-objective","title":"Demonstration Guided Multi-Objective Reinforcement Learning","date":"2024-04-05","arxiv_id":"2404.03997","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-iot-intelligence-a-transformer","title":"Enhancing IoT Intelligence: A Transformer-based Reinforcement Learning Methodology","date":"2024-04-05","arxiv_id":"2404.04205","repositories_listed":0,"syntology":null},{"url":null,"slug":"heterogeneous-multi-agent-reinforcement-2","title":"Heterogeneous Multi-Agent Reinforcement Learning for Zero-Shot Scalable Collaboration","date":"2024-04-05","arxiv_id":"2404.03869","repositories_listed":0,"syntology":null},{"url":null,"slug":"intervention-assisted-policy-gradient-methods","title":"Intervention-Assisted Policy Gradient Methods for Online Stochastic Queuing Network Optimization: Technical Report","date":"2024-04-05","arxiv_id":"2404.04106","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-based-reset-policy","title":"A Reinforcement Learning based Reset Policy for CDCL SAT Solvers","date":"2024-04-04","arxiv_id":"2404.03753","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-small-language-models-help-large-language","title":"Can Small Language Models Help Large Language Models Reason Better?: LM-Guided Chain-of-Thought","date":"2024-04-04","arxiv_id":"2404.03414","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-organized-arrival-system-for-urban-air","title":"Self-organized free-flight arrival for urban air mobility","date":"2024-04-04","arxiv_id":"2404.03710","repositories_listed":0,"syntology":null},{"url":null,"slug":"ad4rl-autonomous-driving-benchmarks-for","title":"AD4RL: Autonomous Driving Benchmarks for Offline Reinforcement Learning with Value-based Dataset","date":"2024-04-03","arxiv_id":"2404.02429","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-traveling","title":"Deep Reinforcement Learning for Traveling Purchaser Problems","date":"2024-04-03","arxiv_id":"2404.02476","repositories_listed":0,"syntology":null},{"url":null,"slug":"methodology-for-interpretable-reinforcement","title":"Methodology for Interpretable Reinforcement Learning for Optimizing Mechanical Ventilation","date":"2024-04-03","arxiv_id":"2404.03105","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-categorical","title":"Reinforcement Learning in Categorical Cybernetics","date":"2024-04-03","arxiv_id":"2404.02688","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-autonomous-swarm-formation-for","title":"Distributed Autonomous Swarm Formation for Dynamic Network Bridging","date":"2024-04-02","arxiv_id":"2404.01557","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergence-of-chemotactic-strategies-with","title":"Emergence of Chemotactic Strategies with Multi-Agent Reinforcement Learning","date":"2024-04-02","arxiv_id":"2404.01999","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-game-a-model-based-and-imitation","title":"Imitation Game: A Model-based and Imitation Learning Deep Reinforcement Learning Hybrid","date":"2024-04-02","arxiv_id":"2404.01794","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-control-camera-exposure-via","title":"Learning to Control Camera Exposure via Reinforcement Learning","date":"2024-04-02","arxiv_id":"2404.01636","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-granular-adversarial-attacks-against","title":"Multi-granular Adversarial Attacks against Black-box Neural Ranking Models","date":"2024-04-02","arxiv_id":"2404.01574","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlrm-vision-language-models-act-as-reward","title":"VLRM: Vision-Language Models act as Reward Models for Image Captioning","date":"2024-04-02","arxiv_id":"2404.01911","repositories_listed":0,"syntology":null},{"url":null,"slug":"camo-correlation-aware-mask-optimization-with","title":"CAMO: Correlation-Aware Mask Optimization with Modulated Reinforcement Learning","date":"2024-04-01","arxiv_id":"2404.00980","repositories_listed":0,"syntology":null},{"url":null,"slug":"mtlight-efficient-multi-task-reinforcement","title":"MTLight: Efficient Multi-Task Reinforcement Learning for Traffic Signal Control","date":"2024-04-01","arxiv_id":"2404.00886","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-mul-multiplier-design-optimization-with","title":"RL-MUL 2.0: Multiplier Design Optimization with Parallel Deep Reinforcement Learning and Space Reduction","date":"2024-03-31","arxiv_id":"2404.00639","repositories_listed":0,"syntology":null},{"url":null,"slug":"facilitating-reinforcement-learning-for","title":"Facilitating Reinforcement Learning for Process Control Using Transfer Learning: Overview and Perspectives","date":"2024-03-30","arxiv_id":"2404.00247","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctrl-sim-reactive-and-controllable-driving","title":"CtRL-Sim: Reactive and Controllable Driving Agents with Offline Reinforcement Learning","date":"2024-03-29","arxiv_id":"2403.19918","repositories_listed":0,"syntology":null},{"url":null,"slug":"removing-the-need-for-ground-truth-uwb-data","title":"Removing the need for ground truth UWB data collection: self-supervised ranging error correction using deep reinforcement learning","date":"2024-03-28","arxiv_id":"2403.19262","repositories_listed":0,"syntology":null},{"url":null,"slug":"cat-constraints-as-terminations-for-legged","title":"CaT: Constraints as Terminations for Legged Locomotion Reinforcement Learning","date":"2024-03-27","arxiv_id":"2403.18765","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-deraining-via-self-supervised","title":"Image Deraining via Self-supervised Reinforcement Learning","date":"2024-03-27","arxiv_id":"2403.18270","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-model-checking-of-stochastic","title":"Probabilistic Model Checking of Stochastic Reinforcement Learning Policies","date":"2024-03-27","arxiv_id":"2403.18725","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-and-robust-reinforcement-learning","title":"Safe and Robust Reinforcement Learning: Principles and Practice","date":"2024-03-27","arxiv_id":"2403.18539","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-of-switchback-designs-in","title":"An Analysis of Switchback Designs in Reinforcement Learning","date":"2024-03-26","arxiv_id":"2403.17285","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-open-source-end-to-end-logic-optimization","title":"An Open-source End-to-End Logic Optimization Framework for Large-scale Boolean Network with Reinforcement Learning","date":"2024-03-26","arxiv_id":"2403.17395","repositories_listed":0,"syntology":null},{"url":null,"slug":"ma4div-multi-agent-reinforcement-learning-for","title":"MA4DIV: Multi-Agent Reinforcement Learning for Search Result Diversification","date":"2024-03-26","arxiv_id":"2403.17421","repositories_listed":0,"syntology":null},{"url":null,"slug":"paths-to-equilibrium-in-normal-form-games","title":"Paths to Equilibrium in Games","date":"2024-03-26","arxiv_id":"2403.18079","repositories_listed":0,"syntology":null},{"url":null,"slug":"prioritized-league-reinforcement-learning-for","title":"Prioritized League Reinforcement Learning for Large-Scale Heterogeneous Multiagent Systems","date":"2024-03-26","arxiv_id":"2403.18057","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-distributional-offline","title":"Uncertainty-aware Distributional Offline Reinforcement Learning","date":"2024-03-26","arxiv_id":"2403.17646","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-and-mean-variance","title":"Deep Reinforcement Learning and Mean-Variance Strategies for Responsible Portfolio Optimization","date":"2024-03-25","arxiv_id":"2403.16667","repositories_listed":0,"syntology":null},{"url":null,"slug":"speeding-up-path-planning-via-reinforcement","title":"Speeding Up Path Planning via Reinforcement Learning in MCTS for Automated Parking","date":"2024-03-25","arxiv_id":"2403.17234","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-modeling-of-deep-reinforcement","title":"Interpretable Modeling of Deep Reinforcement Learning Driven Scheduling","date":"2024-03-24","arxiv_id":"2403.16293","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-feature-selection-for-inverse","title":"Automated Feature Selection for Inverse Reinforcement Learning","date":"2024-03-22","arxiv_id":"2403.15079","repositories_listed":0,"syntology":null},{"url":null,"slug":"subequivariant-reinforcement-learning","title":"Subequivariant Reinforcement Learning Framework for Coordinated Motion Control","date":"2024-03-22","arxiv_id":"2403.15100","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-reinforcement-learning-with-1","title":"Constrained Reinforcement Learning with Smoothed Log Barrier Function","date":"2024-03-21","arxiv_id":"2403.14508","repositories_listed":0,"syntology":null},{"url":null,"slug":"heuristic-algorithm-based-action-masking","title":"Heuristic Algorithm-based Action Masking Reinforcement Learning (HAAM-RL) with Ensemble Inference Method","date":"2024-03-21","arxiv_id":"2403.14110","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-value-tracking-for-deep-reinforcement","title":"Fast Value Tracking for Deep Reinforcement Learning","date":"2024-03-19","arxiv_id":"2403.13178","repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-ingredients-for-offline-reinforcement","title":"Simple Ingredients for Offline Reinforcement Learning","date":"2024-03-19","arxiv_id":"2403.13097","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-statistical-arbitrage-with","title":"Advanced Statistical Arbitrage with Reinforcement Learning","date":"2024-03-18","arxiv_id":"2403.12180","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-multitask-representation-learning-for","title":"Offline Multitask Representation Learning for Reinforcement Learning","date":"2024-03-18","arxiv_id":"2403.11574","repositories_listed":0,"syntology":null},{"url":null,"slug":"pessimistic-causal-reinforcement-learning","title":"Pessimistic Causal Reinforcement Learning with Mediators for Confounded Offline Data","date":"2024-03-18","arxiv_id":"2403.11841","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-generalizable","title":"Reinforcement Learning with Generalizable Gaussian Splatting","date":"2024-03-18","arxiv_id":"2404.07950","repositories_listed":0,"syntology":null},{"url":null,"slug":"supervised-fine-tuning-as-inverse","title":"Supervised Fine-Tuning as Inverse Reinforcement Learning","date":"2024-03-18","arxiv_id":"2403.12017","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-value-of-reward-lookahead-in","title":"The Value of Reward Lookahead in Reinforcement Learning","date":"2024-03-18","arxiv_id":"2403.11637","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveil-conditional-diffusion-models-with","title":"Unveil Conditional Diffusion Models with Classifier-free Guidance: A Sharp Statistical Theory","date":"2024-03-18","arxiv_id":"2403.11968","repositories_listed":0,"syntology":null},{"url":null,"slug":"phasic-diversity-optimization-for-population","title":"Phasic Diversity Optimization for Population-Based Reinforcement Learning","date":"2024-03-17","arxiv_id":"2403.11114","repositories_listed":0,"syntology":null},{"url":null,"slug":"prior-dependent-analysis-of-posterior","title":"Prior-dependent analysis of posterior sampling reinforcement learning with function approximation","date":"2024-03-17","arxiv_id":"2403.11175","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-options","title":"Reinforcement Learning with Options and State Representation","date":"2024-03-16","arxiv_id":"2403.10855","repositories_listed":0,"syntology":null},{"url":null,"slug":"visarl-visual-reinforcement-learning-guided","title":"ViSaRL: Visual Reinforcement Learning Guided by Human Saliency","date":"2024-03-16","arxiv_id":"2403.10940","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-structured-neural-architecture-search","title":"Chain-structured neural architecture search for financial time series forecasting","date":"2024-03-15","arxiv_id":"2403.14695","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-enhanced-reinforcement-learning-for","title":"Graph Enhanced Reinforcement Learning for Effective Group Formation in Collaborative Problem Solving","date":"2024-03-15","arxiv_id":"2403.10006","repositories_listed":0,"syntology":null},{"url":null,"slug":"perl-parameter-efficient-reinforcement","title":"Parameter Efficient Reinforcement Learning from Human Feedback","date":"2024-03-15","arxiv_id":"2403.10704","repositories_listed":0,"syntology":null},{"url":null,"slug":"hrlaif-improvements-in-helpfulness-and","title":"HRLAIF: Improvements in Helpfulness and Harmlessness in Open-domain Reinforcement Learning From AI Feedback","date":"2024-03-13","arxiv_id":"2403.08309","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-operators-for-enabling-parallel-planning","title":"Meta-operators for Enabling Parallel Planning Using Deep Reinforcement Learning","date":"2024-03-13","arxiv_id":"2403.08910","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-optimization-using-adaptive","title":"Multi-Objective Optimization Using Adaptive Distributed Reinforcement Learning","date":"2024-03-13","arxiv_id":"2403.08879","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-improved-strategy-for-blood-glucose","title":"An Improved Strategy for Blood Glucose Control Using Multi-Step Deep Reinforcement Learning","date":"2024-03-12","arxiv_id":"2403.07566","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-agents-dream-of-electric-sheep-improving","title":"Do Agents Dream of Electric Sheep?: Improving Generalization in Reinforcement Learning through Generative Learning","date":"2024-03-12","arxiv_id":"2403.07979","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-reinforcement-learning-from-human-1","title":"Improving Reinforcement Learning from Human Feedback Using Contrastive Rewards","date":"2024-03-12","arxiv_id":"2403.07708","repositories_listed":0,"syntology":null},{"url":null,"slug":"symmetric-q-learning-reducing-skewness-of","title":"Symmetric Q-learning: Reducing Skewness of Bellman Error in Online Reinforcement Learning","date":"2024-03-12","arxiv_id":"2403.07704","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-modelling","title":"Deep Reinforcement Learning for Modelling Protein Complexes","date":"2024-03-11","arxiv_id":"2405.02299","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepsafempc-deep-learning-based-model","title":"DeepSafeMPC: Deep Learning-Based Model Predictive Control for Safe Multi-Agent Reinforcement Learning","date":"2024-03-11","arxiv_id":"2403.06397","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-context-exploration-exploitation-for","title":"In-context Exploration-Exploitation for Reinforcement Learning","date":"2024-03-11","arxiv_id":"2403.06826","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-model-driven-radiology-report","title":"Large Model driven Radiology Report Generation with Clinical Quality Reinforcement Learning","date":"2024-03-11","arxiv_id":"2403.06728","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantifying-the-sensitivity-of-inverse","title":"Quantifying the Sensitivity of Inverse Reinforcement Learning to Misspecification","date":"2024-03-11","arxiv_id":"2403.06854","repositories_listed":0,"syntology":null},{"url":null,"slug":"tactical-decision-making-for-autonomous","title":"Tactical Decision Making for Autonomous Trucks by Deep Reinforcement Learning with Total Cost of Operation Based Reward","date":"2024-03-11","arxiv_id":"2403.06524","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferable-reinforcement-learning-via","title":"Distributional Successor Features Enable Zero-Shot Policy Optimization","date":"2024-03-10","arxiv_id":"2403.06328","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-classification-performance-via","title":"Enhancing Classification Performance via Reinforcement Learning for Feature Selection","date":"2024-03-09","arxiv_id":"2403.05979","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-paycheck-optimization","title":"Reinforcement Learning Paycheck Optimization for Multivariate Financial Goals","date":"2024-03-09","arxiv_id":"2403.06011","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-with-a","title":"Multi-Agent Reinforcement Learning with a Hierarchy of Reward Machines","date":"2024-03-08","arxiv_id":"2403.07005","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-multi-party-reinforcement-learning","title":"Provable Multi-Party Reinforcement Learning with Diverse Human Feedback","date":"2024-03-08","arxiv_id":"2403.05006","repositories_listed":0,"syntology":null},{"url":null,"slug":"shielded-deep-reinforcement-learning-for","title":"Shielded Deep Reinforcement Learning for Complex Spacecraft Tasking","date":"2024-03-08","arxiv_id":"2403.05693","repositories_listed":0,"syntology":null},{"url":null,"slug":"switching-the-loss-reduces-the-cost-in-batch","title":"Switching the Loss Reduces the Cost in Batch Reinforcement Learning","date":"2024-03-08","arxiv_id":"2403.05385","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-mechanism-informed-reinforcement-learning","title":"A mechanism-driven reinforcement learning framework for shape optimization of airfoils","date":"2024-03-07","arxiv_id":"2403.04329","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-algorithm-for-adversarial-linear","title":"Improved Algorithm for Adversarial Linear Mixture MDPs with Bandit Feedback and Unknown Transition","date":"2024-03-07","arxiv_id":"2403.04568","repositories_listed":0,"syntology":null}],"record_sha256":"ffdfe9e0a4ccee8c747453e6de77d6502d9fcd504fcfb1573809d64e3178df6e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}