{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/56","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":56,"pages_in_order":135,"rows_per_page":100,"rows":[5501,5600],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/55","next":"/task/reinforcement-learning-2/papers/57","papers":[{"url":null,"slug":"ad4rl-autonomous-driving-benchmarks-for","title":"AD4RL: Autonomous Driving Benchmarks for Offline Reinforcement Learning with Value-based Dataset","date":"2024-04-03","arxiv_id":"2404.02429","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-traveling","title":"Deep Reinforcement Learning for Traveling Purchaser Problems","date":"2024-04-03","arxiv_id":"2404.02476","repositories_listed":0,"syntology":null},{"url":null,"slug":"methodology-for-interpretable-reinforcement","title":"Methodology for Interpretable Reinforcement Learning for Optimizing Mechanical Ventilation","date":"2024-04-03","arxiv_id":"2404.03105","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-categorical","title":"Reinforcement Learning in Categorical Cybernetics","date":"2024-04-03","arxiv_id":"2404.02688","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-exploration-in-bayesian-model-based","title":"Active Exploration in Bayesian Model-based Reinforcement Learning for Robot Manipulation","date":"2024-04-02","arxiv_id":"2404.01867","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-autonomous-swarm-formation-for","title":"Distributed Autonomous Swarm Formation for Dynamic Network Bridging","date":"2024-04-02","arxiv_id":"2404.01557","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergence-of-chemotactic-strategies-with","title":"Emergence of Chemotactic Strategies with Multi-Agent Reinforcement Learning","date":"2024-04-02","arxiv_id":"2404.01999","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitation-game-a-model-based-and-imitation","title":"Imitation Game: A Model-based and Imitation Learning Deep Reinforcement Learning Hybrid","date":"2024-04-02","arxiv_id":"2404.01794","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-control-camera-exposure-via","title":"Learning to Control Camera Exposure via Reinforcement Learning","date":"2024-04-02","arxiv_id":"2404.01636","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-granular-adversarial-attacks-against","title":"Multi-granular Adversarial Attacks against Black-box Neural Ranking Models","date":"2024-04-02","arxiv_id":"2404.01574","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlrm-vision-language-models-act-as-reward","title":"VLRM: Vision-Language Models act as Reward Models for Image Captioning","date":"2024-04-02","arxiv_id":"2404.01911","repositories_listed":0,"syntology":null},{"url":null,"slug":"camo-correlation-aware-mask-optimization-with","title":"CAMO: Correlation-Aware Mask Optimization with Modulated Reinforcement Learning","date":"2024-04-01","arxiv_id":"2404.00980","repositories_listed":0,"syntology":null},{"url":null,"slug":"mtlight-efficient-multi-task-reinforcement","title":"MTLight: Efficient Multi-Task Reinforcement Learning for Traffic Signal Control","date":"2024-04-01","arxiv_id":"2404.00886","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-mul-multiplier-design-optimization-with","title":"RL-MUL 2.0: Multiplier Design Optimization with Parallel Deep Reinforcement Learning and Space Reduction","date":"2024-03-31","arxiv_id":"2404.00639","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-autoencoders-for-exteroceptive","title":"Variational Autoencoders for exteroceptive perception in reinforcement learning-based collision avoidance","date":"2024-03-31","arxiv_id":"2404.00623","repositories_listed":0,"syntology":null},{"url":null,"slug":"facilitating-reinforcement-learning-for","title":"Facilitating Reinforcement Learning for Process Control Using Transfer Learning: Overview and Perspectives","date":"2024-03-30","arxiv_id":"2404.00247","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctrl-sim-reactive-and-controllable-driving","title":"CtRL-Sim: Reactive and Controllable Driving Agents with Offline Reinforcement Learning","date":"2024-03-29","arxiv_id":"2403.19918","repositories_listed":0,"syntology":null},{"url":null,"slug":"encomp-enhanced-covert-maneuver-planning","title":"EnCoMP: Enhanced Covert Maneuver Planning with Adaptive Threat-Aware Visibility Estimation using Offline Reinforcement Learning","date":"2024-03-29","arxiv_id":"2403.20016","repositories_listed":0,"syntology":null},{"url":null,"slug":"removing-the-need-for-ground-truth-uwb-data","title":"Removing the need for ground truth UWB data collection: self-supervised ranging error correction using deep reinforcement learning","date":"2024-03-28","arxiv_id":"2403.19262","repositories_listed":0,"syntology":null},{"url":null,"slug":"cat-constraints-as-terminations-for-legged","title":"CaT: Constraints as Terminations for Legged Locomotion Reinforcement Learning","date":"2024-03-27","arxiv_id":"2403.18765","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-deraining-via-self-supervised","title":"Image Deraining via Self-supervised Reinforcement Learning","date":"2024-03-27","arxiv_id":"2403.18270","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-model-checking-of-stochastic","title":"Probabilistic Model Checking of Stochastic Reinforcement Learning Policies","date":"2024-03-27","arxiv_id":"2403.18725","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-and-robust-reinforcement-learning","title":"Safe and Robust Reinforcement Learning: Principles and Practice","date":"2024-03-27","arxiv_id":"2403.18539","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-of-switchback-designs-in","title":"An Analysis of Switchback Designs in Reinforcement Learning","date":"2024-03-26","arxiv_id":"2403.17285","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-open-source-end-to-end-logic-optimization","title":"An Open-source End-to-End Logic Optimization Framework for Large-scale Boolean Network with Reinforcement Learning","date":"2024-03-26","arxiv_id":"2403.17395","repositories_listed":0,"syntology":null},{"url":null,"slug":"ma4div-multi-agent-reinforcement-learning-for","title":"MA4DIV: Multi-Agent Reinforcement Learning for Search Result Diversification","date":"2024-03-26","arxiv_id":"2403.17421","repositories_listed":0,"syntology":null},{"url":null,"slug":"paths-to-equilibrium-in-normal-form-games","title":"Paths to Equilibrium in Games","date":"2024-03-26","arxiv_id":"2403.18079","repositories_listed":0,"syntology":null},{"url":null,"slug":"prioritized-league-reinforcement-learning-for","title":"Prioritized League Reinforcement Learning for Large-Scale Heterogeneous Multiagent Systems","date":"2024-03-26","arxiv_id":"2403.18057","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-distributional-offline","title":"Uncertainty-aware Distributional Offline Reinforcement Learning","date":"2024-03-26","arxiv_id":"2403.17646","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-and-mean-variance","title":"Deep Reinforcement Learning and Mean-Variance Strategies for Responsible Portfolio Optimization","date":"2024-03-25","arxiv_id":"2403.16667","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-role-of-state","title":"Offline Reinforcement Learning: Role of State Aggregation and Trajectory Data","date":"2024-03-25","arxiv_id":"2403.17091","repositories_listed":0,"syntology":null},{"url":null,"slug":"speeding-up-path-planning-via-reinforcement","title":"Speeding Up Path Planning via Reinforcement Learning in MCTS for Automated Parking","date":"2024-03-25","arxiv_id":"2403.17234","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-modeling-of-deep-reinforcement","title":"Interpretable Modeling of Deep Reinforcement Learning Driven Scheduling","date":"2024-03-24","arxiv_id":"2403.16293","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-feature-selection-for-inverse","title":"Automated Feature Selection for Inverse Reinforcement Learning","date":"2024-03-22","arxiv_id":"2403.15079","repositories_listed":0,"syntology":null},{"url":null,"slug":"subequivariant-reinforcement-learning","title":"Subequivariant Reinforcement Learning Framework for Coordinated Motion Control","date":"2024-03-22","arxiv_id":"2403.15100","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-reinforcement-learning-with-1","title":"Constrained Reinforcement Learning with Smoothed Log Barrier Function","date":"2024-03-21","arxiv_id":"2403.14508","repositories_listed":0,"syntology":null},{"url":null,"slug":"heuristic-algorithm-based-action-masking","title":"Heuristic Algorithm-based Action Masking Reinforcement Learning (HAAM-RL) with Ensemble Inference Method","date":"2024-03-21","arxiv_id":"2403.14110","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-continuity-and-smoothness-of-the-value","title":"On the continuity and smoothness of the value function in reinforcement learning and optimal control","date":"2024-03-21","arxiv_id":"2403.14432","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-model-based-reinforcement-learning","title":"Robust Model Based Reinforcement Learning Using $\\mathcal{L}_1$ Adaptive Control","date":"2024-03-21","arxiv_id":"2403.14860","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-value-tracking-for-deep-reinforcement","title":"Fast Value Tracking for Deep Reinforcement Learning","date":"2024-03-19","arxiv_id":"2403.13178","repositories_listed":0,"syntology":null},{"url":null,"slug":"simple-ingredients-for-offline-reinforcement","title":"Simple Ingredients for Offline Reinforcement Learning","date":"2024-03-19","arxiv_id":"2403.13097","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-halpern-iteration-in-normed-spaces","title":"Stochastic Halpern iteration in normed spaces and applications to reinforcement learning","date":"2024-03-19","arxiv_id":"2403.12338","repositories_listed":0,"syntology":null},{"url":null,"slug":"advanced-statistical-arbitrage-with","title":"Advanced Statistical Arbitrage with Reinforcement Learning","date":"2024-03-18","arxiv_id":"2403.12180","repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrapping-reinforcement-learning-with","title":"Bootstrapping Reinforcement Learning with Imitation for Vision-Based Agile Flight","date":"2024-03-18","arxiv_id":"2403.12203","repositories_listed":0,"syntology":null},{"url":null,"slug":"demystifying-deep-reinforcement-learning","title":"Demystifying the Physics of Deep Reinforcement Learning-Based Autonomous Vehicle Decision-Making","date":"2024-03-18","arxiv_id":"2403.11432","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-reinforcement-learning-based-home","title":"Explainable Reinforcement Learning-based Home Energy Management Systems using Differentiable Decision Trees","date":"2024-03-18","arxiv_id":"2403.11947","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-multitask-representation-learning-for","title":"Offline Multitask Representation Learning for Reinforcement Learning","date":"2024-03-18","arxiv_id":"2403.11574","repositories_listed":0,"syntology":null},{"url":null,"slug":"pessimistic-causal-reinforcement-learning","title":"Pessimistic Causal Reinforcement Learning with Mediators for Confounded Offline Data","date":"2024-03-18","arxiv_id":"2403.11841","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-generalizable","title":"Reinforcement Learning with Generalizable Gaussian Splatting","date":"2024-03-18","arxiv_id":"2404.07950","repositories_listed":0,"syntology":null},{"url":null,"slug":"supervised-fine-tuning-as-inverse","title":"Supervised Fine-Tuning as Inverse Reinforcement Learning","date":"2024-03-18","arxiv_id":"2403.12017","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-value-of-reward-lookahead-in","title":"The Value of Reward Lookahead in Reinforcement Learning","date":"2024-03-18","arxiv_id":"2403.11637","repositories_listed":0,"syntology":null},{"url":null,"slug":"unveil-conditional-diffusion-models-with","title":"Unveil Conditional Diffusion Models with Classifier-free Guidance: A Sharp Statistical Theory","date":"2024-03-18","arxiv_id":"2403.11968","repositories_listed":0,"syntology":null},{"url":null,"slug":"phasic-diversity-optimization-for-population","title":"Phasic Diversity Optimization for Population-Based Reinforcement Learning","date":"2024-03-17","arxiv_id":"2403.11114","repositories_listed":0,"syntology":null},{"url":null,"slug":"prior-dependent-analysis-of-posterior","title":"Prior-dependent analysis of posterior sampling reinforcement learning with function approximation","date":"2024-03-17","arxiv_id":"2403.11175","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-scalable-and-parallelizable-digital-twin","title":"Mixed-Reality Digital Twins: Leveraging the Physical and Virtual Worlds for Hybrid Sim2Real Transition of Multi-Agent Reinforcement Learning Policies","date":"2024-03-16","arxiv_id":"2403.10996","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-options","title":"Reinforcement Learning with Options and State Representation","date":"2024-03-16","arxiv_id":"2403.10855","repositories_listed":0,"syntology":null},{"url":null,"slug":"visarl-visual-reinforcement-learning-guided","title":"ViSaRL: Visual Reinforcement Learning Guided by Human Saliency","date":"2024-03-16","arxiv_id":"2403.10940","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-structured-neural-architecture-search","title":"Chain-structured neural architecture search for financial time series forecasting","date":"2024-03-15","arxiv_id":"2403.14695","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-enhanced-reinforcement-learning-for","title":"Graph Enhanced Reinforcement Learning for Effective Group Formation in Collaborative Problem Solving","date":"2024-03-15","arxiv_id":"2403.10006","repositories_listed":0,"syntology":null},{"url":null,"slug":"perl-parameter-efficient-reinforcement","title":"Parameter Efficient Reinforcement Learning from Human Feedback","date":"2024-03-15","arxiv_id":"2403.10704","repositories_listed":0,"syntology":null},{"url":null,"slug":"hrlaif-improvements-in-helpfulness-and","title":"HRLAIF: Improvements in Helpfulness and Harmlessness in Open-domain Reinforcement Learning From AI Feedback","date":"2024-03-13","arxiv_id":"2403.08309","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-operators-for-enabling-parallel-planning","title":"Meta-operators for Enabling Parallel Planning Using Deep Reinforcement Learning","date":"2024-03-13","arxiv_id":"2403.08910","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-objective-optimization-using-adaptive","title":"Multi-Objective Optimization Using Adaptive Distributed Reinforcement Learning","date":"2024-03-13","arxiv_id":"2403.08879","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-improved-strategy-for-blood-glucose","title":"An Improved Strategy for Blood Glucose Control Using Multi-Step Deep Reinforcement Learning","date":"2024-03-12","arxiv_id":"2403.07566","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-optimal-fuel-consumption-of-hev-a","title":"Constrained Optimal Fuel Consumption of HEV: A Constrained Reinforcement Learning Approach","date":"2024-03-12","arxiv_id":"2403.07503","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-agents-dream-of-electric-sheep-improving","title":"Do Agents Dream of Electric Sheep?: Improving Generalization in Reinforcement Learning through Generative Learning","date":"2024-03-12","arxiv_id":"2403.07979","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-reinforcement-learning-from-human-1","title":"Improving Reinforcement Learning from Human Feedback Using Contrastive Rewards","date":"2024-03-12","arxiv_id":"2403.07708","repositories_listed":0,"syntology":null},{"url":null,"slug":"symmetric-q-learning-reducing-skewness-of","title":"Symmetric Q-learning: Reducing Skewness of Bellman Error in Online Reinforcement Learning","date":"2024-03-12","arxiv_id":"2403.07704","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-modelling","title":"Deep Reinforcement Learning for Modelling Protein Complexes","date":"2024-03-11","arxiv_id":"2405.02299","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepsafempc-deep-learning-based-model","title":"DeepSafeMPC: Deep Learning-Based Model Predictive Control for Safe Multi-Agent Reinforcement Learning","date":"2024-03-11","arxiv_id":"2403.06397","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-image-caption-generation-using","title":"Enhancing Image Caption Generation Using Reinforcement Learning with Human Feedback","date":"2024-03-11","arxiv_id":"2403.06735","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-context-exploration-exploitation-for","title":"In-context Exploration-Exploitation for Reinforcement Learning","date":"2024-03-11","arxiv_id":"2403.06826","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-model-driven-radiology-report","title":"Large Model driven Radiology Report Generation with Clinical Quality Reinforcement Learning","date":"2024-03-11","arxiv_id":"2403.06728","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathbf-n-k-puzzle-a-cost-efficient-testbed","title":"$\\mathbf{(N,K)}$-Puzzle: A Cost-Efficient Testbed for Benchmarking Reinforcement Learning Algorithms in Generative Language Model","date":"2024-03-11","arxiv_id":"2403.07191","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantifying-the-sensitivity-of-inverse","title":"Quantifying the Sensitivity of Inverse Reinforcement Learning to Misspecification","date":"2024-03-11","arxiv_id":"2403.06854","repositories_listed":0,"syntology":null},{"url":null,"slug":"tactical-decision-making-for-autonomous","title":"Tactical Decision Making for Autonomous Trucks by Deep Reinforcement Learning with Total Cost of Operation Based Reward","date":"2024-03-11","arxiv_id":"2403.06524","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferable-reinforcement-learning-via","title":"Distributional Successor Features Enable Zero-Shot Policy Optimization","date":"2024-03-10","arxiv_id":"2403.06328","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-classification-performance-via","title":"Enhancing Classification Performance via Reinforcement Learning for Feature Selection","date":"2024-03-09","arxiv_id":"2403.05979","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-paycheck-optimization","title":"Reinforcement Learning Paycheck Optimization for Multivariate Financial Goals","date":"2024-03-09","arxiv_id":"2403.06011","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-with-a","title":"Multi-Agent Reinforcement Learning with a Hierarchy of Reward Machines","date":"2024-03-08","arxiv_id":"2403.07005","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-multi-party-reinforcement-learning","title":"Provable Multi-Party Reinforcement Learning with Diverse Human Feedback","date":"2024-03-08","arxiv_id":"2403.05006","repositories_listed":0,"syntology":null},{"url":null,"slug":"shielded-deep-reinforcement-learning-for","title":"Shielded Deep Reinforcement Learning for Complex Spacecraft Tasking","date":"2024-03-08","arxiv_id":"2403.05693","repositories_listed":0,"syntology":null},{"url":null,"slug":"switching-the-loss-reduces-the-cost-in-batch","title":"Switching the Loss Reduces the Cost in Batch Reinforcement Learning","date":"2024-03-08","arxiv_id":"2403.05385","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-mechanism-informed-reinforcement-learning","title":"A mechanism-driven reinforcement learning framework for shape optimization of airfoils","date":"2024-03-07","arxiv_id":"2403.04329","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-algorithm-for-adversarial-linear","title":"Improved Algorithm for Adversarial Linear Mixture MDPs with Bandit Feedback and Unknown Transition","date":"2024-03-07","arxiv_id":"2403.04568","repositories_listed":0,"syntology":null},{"url":null,"slug":"proxy-rlhf-decoupling-generation-and","title":"Proxy-RLHF: Decoupling Generation and Alignment in Large Language Model with Proxy","date":"2024-03-07","arxiv_id":"2403.04283","repositories_listed":0,"syntology":null},{"url":null,"slug":"teaching-large-language-models-to-reason-with","title":"Teaching Large Language Models to Reason with Reinforcement Learning","date":"2024-03-07","arxiv_id":"2403.04642","repositories_listed":0,"syntology":null},{"url":null,"slug":"vlearn-off-policy-learning-with-efficient","title":"Vlearn: Off-Policy Learning with Efficient State-Value Function Estimation","date":"2024-03-07","arxiv_id":"2403.04453","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-online-reinforcement-learning-is-causal","title":"Why Online Reinforcement Learning is Causal","date":"2024-03-07","arxiv_id":"2403.04221","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-applications-of-reinforcement","title":"A Survey on Applications of Reinforcement Learning in Spatial Resource Allocation","date":"2024-03-06","arxiv_id":"2403.03643","repositories_listed":0,"syntology":null},{"url":null,"slug":"dexterous-legged-locomotion-in-confined-3d","title":"Dexterous Legged Locomotion in Confined 3D Spaces with Reinforcement Learning","date":"2024-03-06","arxiv_id":"2403.03848","repositories_listed":0,"syntology":null},{"url":null,"slug":"reconciling-reality-through-simulation-a-real","title":"Reconciling Reality through Simulation: A Real-to-Sim-to-Real Approach for Robust Manipulation","date":"2024-03-06","arxiv_id":"2403.03949","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-zero-shot-reinforcement-learning-strategy","title":"A Zero-Shot Reinforcement Learning Strategy for Autonomous Guidewire Navigation","date":"2024-03-05","arxiv_id":"2403.02777","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-vehicle-decision-and-control","title":"Autonomous vehicle decision and control through reinforcement learning with traffic flow randomization","date":"2024-03-05","arxiv_id":"2403.02882","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-llm-safety-via-constrained-direct","title":"Enhancing LLM Safety via Constrained Direct Preference Optimization","date":"2024-03-04","arxiv_id":"2403.02475","repositories_listed":0,"syntology":null},{"url":null,"slug":"koopman-assisted-reinforcement-learning","title":"Koopman-Assisted Reinforcement Learning","date":"2024-03-04","arxiv_id":"2403.02290","repositories_listed":0,"syntology":null},{"url":null,"slug":"smaug-a-sliding-multidimensional-task-window","title":"SMAUG: A Sliding Multidimensional Task Window-Based MARL Framework for Adaptive Real-Time Subtask Recognition","date":"2024-03-04","arxiv_id":"2403.01816","repositories_listed":0,"syntology":null},{"url":null,"slug":"tsallis-entropy-regularization-for-linearly","title":"Tsallis Entropy Regularization for Linearly Solvable MDP and Linear Quadratic Regulator","date":"2024-03-04","arxiv_id":"2403.01805","repositories_listed":0,"syntology":null},{"url":null,"slug":"twisting-lids-off-with-two-hands","title":"Twisting Lids Off with Two Hands","date":"2024-03-04","arxiv_id":"2403.02338","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-myopic-exploration-through","title":"Sample Efficient Myopic Exploration Through Multitask Reinforcement Learning with Diverse Tasks","date":"2024-03-03","arxiv_id":"2403.01636","repositories_listed":0,"syntology":null}],"record_sha256":"4278cf98be7341f592b4893088e1612839843718a7836fd85e76cb328a6e5d8a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}