{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/62","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":62,"pages_in_order":152,"rows_per_page":100,"rows":[6101,6200],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/61","next":"/task/reinforcement-learning-1/papers/63","papers":[{"url":null,"slug":"by-fair-means-or-foul-quantifying-collusion","title":"By Fair Means or Foul: Quantifying Collusion in a Market Simulation with Deep Reinforcement Learning","date":"2024-06-04","arxiv_id":"2406.02650","repositories_listed":0,"syntology":null},{"url":null,"slug":"fightladder-a-benchmark-for-competitive-multi","title":"FightLadder: A Benchmark for Competitive Multi-Agent Reinforcement Learning","date":"2024-06-04","arxiv_id":"2406.02081","repositories_listed":0,"syntology":null},{"url":null,"slug":"iqrl-implicitly-quantized-representations-for","title":"iQRL -- Implicitly Quantized Representations for Sample-efficient Reinforcement Learning","date":"2024-06-04","arxiv_id":"2406.02696","repositories_listed":0,"syntology":null},{"url":null,"slug":"rectifying-reinforcement-learning-for-reward","title":"Rectifying Reinforcement Learning for Reward Matching","date":"2024-06-04","arxiv_id":"2406.02213","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-lookahead","title":"Reinforcement Learning with Lookahead Information","date":"2024-06-04","arxiv_id":"2406.02258","repositories_listed":0,"syntology":null},{"url":null,"slug":"smaller-batches-bigger-gains-investigating","title":"Smaller Batches, Bigger Gains? Investigating the Impact of Batch Sizes on Reinforcement Learning Based Real-World Production Scheduling","date":"2024-06-04","arxiv_id":"2406.02294","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-theory-of-learnability-for-offline-decision","title":"A Fast Convergence Theory for Offline Decision Making","date":"2024-06-03","arxiv_id":"2406.01378","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-prompting-model-based-offline","title":"Causal prompting model-based offline reinforcement learning","date":"2024-06-03","arxiv_id":"2406.01065","repositories_listed":0,"syntology":null},{"url":null,"slug":"combinatorial-multivariant-multi-armed","title":"Combinatorial Multivariant Multi-Armed Bandits with Applications to Episodic Reinforcement Learning and Beyond","date":"2024-06-03","arxiv_id":"2406.01386","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-learning-based-collaborative","title":"Federated Learning-based Collaborative Wideband Spectrum Sensing and Scheduling for UAVs in UTM Systems","date":"2024-06-03","arxiv_id":"2406.01727","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-the-target-network-in-function-space","title":"Learning the Target Network in Function Space","date":"2024-06-03","arxiv_id":"2406.01838","repositories_listed":0,"syntology":null},{"url":null,"slug":"mot-a-mixture-of-actors-reinforcement","title":"MOT: A Mixture of Actors Reinforcement Learning Method by Optimal Transport for Algorithmic Trading","date":"2024-06-03","arxiv_id":"2407.01577","repositories_listed":0,"syntology":null},{"url":null,"slug":"neorl-efficient-exploration-for-nonepisodic","title":"NeoRL: Efficient Exploration for Nonepisodic RL","date":"2024-06-03","arxiv_id":"2406.01175","repositories_listed":0,"syntology":null},{"url":null,"slug":"recent-trends-in-insect-and-robot-navigation","title":"Reinforcement Learning as a Robotics-Inspired Framework for Insect Navigation: From Spatial Representations to Neural Implementation","date":"2024-06-03","arxiv_id":"2406.01501","repositories_listed":0,"syntology":null},{"url":null,"slug":"revolve-reward-evolution-with-large-language","title":"REvolve: Reward Evolution with Large Language Models using Human Feedback","date":"2024-06-03","arxiv_id":"2406.01309","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-preference-fine-tuning-through","title":"The Importance of Online Data: Understanding Preference Fine-tuning via Coverage","date":"2024-06-03","arxiv_id":"2406.01462","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-digital-twin-framework-for-reinforcement","title":"A Digital Twin Framework for Reinforcement Learning with Real-Time Self-Improvement via Human Assistive Teleoperation","date":"2024-06-02","arxiv_id":"2406.00732","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-predictive-control-and-reinforcement-1","title":"Model Predictive Control and Reinforcement Learning: A Unified Framework Based on Dynamic Programming","date":"2024-06-02","arxiv_id":"2406.00592","repositories_listed":0,"syntology":null},{"url":null,"slug":"decision-mamba-reinforcement-learning-via-1","title":"Decision Mamba: Reinforcement Learning via Hybrid Selective Sequence Modeling","date":"2024-05-31","arxiv_id":"2406.00079","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-sociohydrology","title":"Reinforcement Learning for Sociohydrology","date":"2024-05-31","arxiv_id":"2405.20772","repositories_listed":0,"syntology":null},{"url":null,"slug":"bilevel-reinforcement-learning-via-the","title":"Bilevel reinforcement learning via the development of hyper-gradient without lower-level convexity","date":"2024-05-30","arxiv_id":"2405.19697","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-stimuli-generation-using","title":"Efficient Stimuli Generation using Reinforcement Learning in Design Verification","date":"2024-05-30","arxiv_id":"2405.19815","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-words-to-actions-unveiling-the","title":"From Words to Actions: Unveiling the Theoretical Underpinnings of LLM-Driven Autonomous Systems","date":"2024-05-30","arxiv_id":"2405.19883","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-reinforcement-learning-framework-for","title":"Hybrid Reinforcement Learning Framework for Mixed-Variable Problems","date":"2024-05-30","arxiv_id":"2405.20500","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-discuss-strategically-a-case","title":"Learning to Discuss Strategically: A Case Study on One Night Ultimate Werewolf","date":"2024-05-30","arxiv_id":"2405.19946","repositories_listed":0,"syntology":null},{"url":null,"slug":"sleepernets-universal-backdoor-poisoning","title":"SleeperNets: Universal Backdoor Poisoning Attacks Against Reinforcement Learning Agents","date":"2024-05-30","arxiv_id":"2405.20539","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-discretization-based-non-episodic","title":"Policy Zooming: Adaptive Discretization-based Infinite-Horizon Average-Reward Reinforcement Learning","date":"2024-05-29","arxiv_id":"2405.18793","repositories_listed":0,"syntology":null},{"url":null,"slug":"preferred-action-optimized-diffusion-policies","title":"Preferred-Action-Optimized Diffusion Policies for Offline Reinforcement Learning","date":"2024-05-29","arxiv_id":"2405.18729","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-through-permissibility-shield","title":"Safety through Permissibility: Shield Construction for Fast and Safe Reinforcement Learning","date":"2024-05-29","arxiv_id":"2405.19414","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-incentivized-preference-optimization-a","title":"Value-Incentivized Preference Optimization: A Unified Approach to Online and Offline RLHF","date":"2024-05-29","arxiv_id":"2405.19320","repositories_listed":0,"syntology":null},{"url":null,"slug":"extreme-value-monte-carlo-tree-search","title":"Extreme Value Monte Carlo Tree Search","date":"2024-05-28","arxiv_id":"2405.18248","repositories_listed":0,"syntology":null},{"url":null,"slug":"highway-reinforcement-learning","title":"Highway Reinforcement Learning","date":"2024-05-28","arxiv_id":"2405.18289","repositories_listed":0,"syntology":null},{"url":null,"slug":"mollification-effects-of-policy-gradient","title":"Mollification Effects of Policy Gradient Methods","date":"2024-05-28","arxiv_id":"2405.17832","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-pruning-for-backdoor-mitigation-an","title":"Rethinking Pruning for Backdoor Mitigation: An Optimization Perspective","date":"2024-05-28","arxiv_id":"2405.17746","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-llms-to-better-self-debug-and","title":"LeDex: Training LLMs to Better Self-Debug and Explain Code","date":"2024-05-28","arxiv_id":"2405.18649","repositories_listed":0,"syntology":null},{"url":null,"slug":"biological-neurons-compete-with-deep","title":"Biological Neurons Compete with Deep Reinforcement Learning in Sample Efficiency in a Simulated Gameworld","date":"2024-05-27","arxiv_id":"2405.16946","repositories_listed":0,"syntology":null},{"url":null,"slug":"ontology-enhanced-decision-making-for","title":"Ontology-Enhanced Decision-Making for Autonomous Agents in Dynamic and Partially Observable Environments","date":"2024-05-27","arxiv_id":"2405.17691","repositories_listed":0,"syntology":null},{"url":null,"slug":"oracle-efficient-reinforcement-learning-for","title":"Oracle-Efficient Reinforcement Learning for Max Value Ensembles","date":"2024-05-27","arxiv_id":"2405.16739","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-graph-network-for-constrained","title":"Structured Graph Network for Constrained Robot Crowd Navigation with Low Fidelity Simulation","date":"2024-05-27","arxiv_id":"2405.16830","repositories_listed":0,"syntology":null},{"url":null,"slug":"trajectory-data-suffices-for-statistically","title":"Trajectory Data Suffices for Statistically Efficient Learning in Offline RL with Linear $q^π$-Realizability and Concentrability","date":"2024-05-27","arxiv_id":"2405.16809","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-evolutionary-framework-for-connect-4-as","title":"An Evolutionary Framework for Connect-4 as Test-Bed for Comparison of Advanced Minimax, Q-Learning and MCTS","date":"2024-05-26","arxiv_id":"2405.16595","repositories_listed":0,"syntology":null},{"url":null,"slug":"pick-up-the-pace-a-parameter-free-optimizer","title":"Fast TRAC: A Parameter-Free Optimizer for Lifelong Reinforcement Learning","date":"2024-05-26","arxiv_id":"2405.16642","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-jump-diffusions","title":"Reinforcement Learning for Jump-Diffusions, with Financial Applications","date":"2024-05-26","arxiv_id":"2405.16449","repositories_listed":0,"syntology":null},{"url":"/paper/adaptive-q-network-on-the-fly-target","slug":"adaptive-q-network-on-the-fly-target","title":"Adaptive $Q$-Network: On-the-fly Target Selection for Deep Reinforcement Learning","date":"2024-05-25","arxiv_id":"2405.16195","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/adaptive-q-network-on-the-fly-target#ran","syntology_url":"https://syntology.ai/paper/2405.16195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16195"}},"official":null}},{"url":null,"slug":"aigb-generative-auto-bidding-via-diffusion","title":"AIGB: Generative Auto-bidding via Conditional Diffusion Modeling","date":"2024-05-25","arxiv_id":"2405.16141","repositories_listed":0,"syntology":null},{"url":"/paper/constrained-ensemble-exploration-for","slug":"constrained-ensemble-exploration-for","title":"Constrained Ensemble Exploration for Unsupervised Skill Discovery","date":"2024-05-25","arxiv_id":"2405.16030","repositories_listed":0,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/constrained-ensemble-exploration-for#ran","syntology_url":"https://syntology.ai/paper/2405.16030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16030"}},"official":null}},{"url":null,"slug":"cooperative-backdoor-attack-in-decentralized","title":"Cooperative Backdoor Attack in Decentralized Reinforcement Learning with Theoretical Guarantee","date":"2024-05-24","arxiv_id":"2405.15245","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforcement-learning-via-large","title":"Extracting Heuristics from Large Language Models for Reward Shaping in Reinforcement Learning","date":"2024-05-24","arxiv_id":"2405.15194","repositories_listed":0,"syntology":null},{"url":null,"slug":"embedding-aligned-language-models","title":"Embedding-Aligned Language Models","date":"2024-05-24","arxiv_id":"2406.00024","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-in-the-loop-reinforcement-learning-for","title":"Human-in-the-loop Reinforcement Learning for Data Quality Monitoring in Particle Physics Experiments","date":"2024-05-24","arxiv_id":"2405.15508","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-informed-auto-penetration-testing","title":"Knowledge-Informed Auto-Penetration Testing Based on Reinforcement Learning with Reward Machine","date":"2024-05-24","arxiv_id":"2405.15908","repositories_listed":0,"syntology":null},{"url":null,"slug":"sf-dqn-provable-knowledge-transfer-using","title":"SF-DQN: Provable Knowledge Transfer using Successor Feature for Deep Reinforcement Learning","date":"2024-05-24","arxiv_id":"2405.15920","repositories_listed":0,"syntology":null},{"url":null,"slug":"trojanforge-adversarial-hardware-trojan","title":"TrojanForge: Generating Adversarial Hardware Trojan Examples Using Reinforcement Learning","date":"2024-05-24","arxiv_id":"2405.15184","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-finite-time-analysis-of-distributed-q","title":"A finite time analysis of distributed Q-learning","date":"2024-05-23","arxiv_id":"2405.14078","repositories_listed":0,"syntology":null},{"url":null,"slug":"blood-glucose-control-via-pre-trained","title":"Blood Glucose Control Via Pre-trained Counterfactual Invertible Neural Networks","date":"2024-05-23","arxiv_id":"2405.17458","repositories_listed":0,"syntology":null},{"url":null,"slug":"exclusively-penalized-q-learning-for-offline","title":"Exclusively Penalized Q-learning for Offline Reinforcement Learning","date":"2024-05-23","arxiv_id":"2405.14082","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-methods-for-risk-sensitive","title":"Policy Gradient Methods for Risk-Sensitive Distributional Reinforcement Learning with Provable Convergence","date":"2024-05-23","arxiv_id":"2405.14749","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-stage-ml-guided-decision-rules-for","title":"Efficiently Training Deep-Learning Parametric Policies using Lagrangian Duality","date":"2024-05-23","arxiv_id":"2405.14973","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-algorithm-for-training-autonomous","title":"Autonomous Algorithm for Training Autonomous Vehicles with Minimal Human Intervention","date":"2024-05-22","arxiv_id":"2405.13345","repositories_listed":0,"syntology":null},{"url":null,"slug":"highwayllm-decision-making-and-navigation-in","title":"HighwayLLM: Decision-Making and Navigation in Highway Driving with RL-Informed Language Model","date":"2024-05-22","arxiv_id":"2405.13547","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-llms-assisted-wireless","title":"Large Language Models (LLMs) Assisted Wireless Network Deployment in Urban Settings","date":"2024-05-22","arxiv_id":"2405.13356","repositories_listed":0,"syntology":null},{"url":null,"slug":"leader-reward-for-pomo-based-neural","title":"Leader Reward for POMO-Based Neural Combinatorial Optimization","date":"2024-05-22","arxiv_id":"2405.13947","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multimodal-learning-based-approach-for","title":"A Multimodal Learning-based Approach for Autonomous Landing of UAV","date":"2024-05-21","arxiv_id":"2405.12681","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-with-8","title":"Multi-Agent Reinforcement Learning with Hierarchical Coordination for Emergency Responder Stationing","date":"2024-05-21","arxiv_id":"2405.13205","repositories_listed":0,"syntology":null},{"url":null,"slug":"practical-and-efficient-quantum-circuit","title":"Practical and efficient quantum circuit synthesis and transpiling with Reinforcement Learning","date":"2024-05-21","arxiv_id":"2405.13196","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-robustness-assessment-adversarial","title":"Rethinking Robustness Assessment: Adversarial Attacks on Learning-based Quadrupedal Locomotion Controllers","date":"2024-05-21","arxiv_id":"2405.12424","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-impact-of-choice-on-deep","title":"Investigating the Impact of Choice on Deep Reinforcement Learning for Space Controls","date":"2024-05-20","arxiv_id":"2405.12355","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-future-representation-with-synthetic","title":"Learning Future Representation with Synthetic Observations for Sample-efficient Reinforcement Learning","date":"2024-05-20","arxiv_id":"2405.11740","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparisons-are-all-you-need-for-optimizing","title":"Comparisons Are All You Need for Optimizing Smooth Functions","date":"2024-05-19","arxiv_id":"2405.11454","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-no-harm-a-counterfactual-approach-to-safe","title":"Do No Harm: A Counterfactual Approach to Safe Reinforcement Learning","date":"2024-05-19","arxiv_id":"2405.11669","repositories_listed":0,"syntology":null},{"url":null,"slug":"combined-film-and-pulse-heating-of-lithium","title":"Combined film and pulse heating of lithium ion batteries to improve performance in low ambient temperature","date":"2024-05-18","arxiv_id":"2405.11388","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-control-barrier-functions-for-rl","title":"Optimal control barrier functions for RL based safe powertrain control","date":"2024-05-18","arxiv_id":"2405.11391","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-policy-enhancing-offline","title":"Towards Robust Policy: Enhancing Offline Reinforcement Learning with Adversarial Attacks and Defenses","date":"2024-05-18","arxiv_id":"2405.11206","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-based-multi-agent-reinforcement-learning","title":"LLM-based Multi-Agent Reinforcement Learning: Current and Future Directions","date":"2024-05-17","arxiv_id":"2405.11106","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-tuning-large-vision-language-models-as","title":"Fine-Tuning Large Vision-Language Models as Decision-Making Agents via Reinforcement Learning","date":"2024-05-16","arxiv_id":"2405.10292","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-q-learning-for-large-discrete","title":"Stochastic Q-learning for Large Discrete Action Spaces","date":"2024-05-16","arxiv_id":"2405.10310","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-in-earthquake-engineering-a","title":"Deep Learning in Earthquake Engineering: A Comprehensive Review","date":"2024-05-15","arxiv_id":"2405.09021","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-two-time-scale-stochastic-gradient","title":"Fast Two-Time-Scale Stochastic Gradient Method with Applications in Reinforcement Learning","date":"2024-05-15","arxiv_id":"2405.09660","repositories_listed":0,"syntology":null},{"url":null,"slug":"im-rag-multi-round-retrieval-augmented","title":"IM-RAG: Multi-Round Retrieval-Augmented Generation Through Learning Inner Monologues","date":"2024-05-15","arxiv_id":"2405.13021","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-real-time-2","title":"Deep Reinforcement Learning for Real-Time Ground Delay Program Revision and Corresponding Flight Delay Assignments","date":"2024-05-14","arxiv_id":"2405.08298","repositories_listed":0,"syntology":null},{"url":null,"slug":"vmfer-von-mises-fisher-experience-resampling","title":"vMFER: Von Mises-Fisher Experience Resampling Based on Uncertainty of Gradient Directions for Policy Improvement","date":"2024-05-14","arxiv_id":"2405.08638","repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsic-rewards-for-exploration-without","title":"Intrinsic Rewards for Exploration without Harm from Observational Noise: A Simulation Study Based on the Free Energy Principle","date":"2024-05-13","arxiv_id":"2405.07473","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-regret-in-linear-mdps-with","title":"Near-Optimal Regret in Linear MDPs with Aggregate Bandit Feedback","date":"2024-05-13","arxiv_id":"2405.07637","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-network-compression-for-reinforcement","title":"Neural Network Compression for Reinforcement Learning Tasks","date":"2024-05-13","arxiv_id":"2405.07748","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-risk-for-assistive-reinforcement","title":"Reducing Risk for Assistive Reinforcement Learning Policies with Diffusion Models","date":"2024-05-13","arxiv_id":"2405.07603","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-successor-representations-for-task","title":"Ensemble Successor Representations for Task Generalization in Offline-to-Online Reinforcement Learning","date":"2024-05-12","arxiv_id":"2405.07223","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairness-in-reinforcement-learning-a-survey","title":"Fairness in Reinforcement Learning: A Survey","date":"2024-05-11","arxiv_id":"2405.06909","repositories_listed":0,"syntology":null},{"url":null,"slug":"dominion-a-new-frontier-for-ai-research","title":"Dominion: A New Frontier for AI Research","date":"2024-05-10","arxiv_id":"2405.06846","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-targeted-molecule-generation","title":"Improving Targeted Molecule Generation through Language Model Fine-Tuning Via Reinforcement Learning","date":"2024-05-10","arxiv_id":"2405.06836","repositories_listed":0,"syntology":null},{"url":null,"slug":"space-processor-computation-time-analysis-for","title":"Space Processor Computation Time Analysis for Reinforcement Learning and Run Time Assurance Control Policies","date":"2024-05-10","arxiv_id":"2405.06771","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-overview-of-machine-learning-enabled-1","title":"An Overview of Machine Learning-Enabled Optimization for Reconfigurable Intelligent Surfaces-Aided 6G Networks: From Reinforcement Learning to Large Language Models","date":"2024-05-09","arxiv_id":"2405.17439","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-stochastic-policy-gradient-negative","title":"Fast Stochastic Policy Gradient: Negative Momentum for Reinforcement Learning","date":"2024-05-08","arxiv_id":"2405.12228","repositories_listed":0,"syntology":null},{"url":null,"slug":"genetic-drift-regularization-on-preventing","title":"Genetic Drift Regularization: on preventing Actor Injection from breaking Evolution Strategies","date":"2024-05-07","arxiv_id":"2405.04322","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-offline-reinforcement-learning-with","title":"Improving Offline Reinforcement Learning with Inaccurate Simulators","date":"2024-05-07","arxiv_id":"2405.04307","repositories_listed":0,"syntology":null},{"url":null,"slug":"roadside-units-assisted-localized-automated","title":"Roadside Units Assisted Localized Automated Vehicle Maneuvering: An Offline Reinforcement Learning Approach","date":"2024-05-07","arxiv_id":"2405.03935","repositories_listed":0,"syntology":null},{"url":null,"slug":"out-of-distribution-adaptation-in-offline-rl","title":"Out-of-Distribution Adaptation in Offline RL: Counterfactual Reasoning via Causal Normalizing Flows","date":"2024-05-06","arxiv_id":"2405.03892","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-learned-non","title":"Safe Reinforcement Learning with Learned Non-Markovian Safety Constraints","date":"2024-05-05","arxiv_id":"2405.03005","repositories_listed":0,"syntology":null},{"url":null,"slug":"uduc-an-uncertainty-driven-approach-for","title":"UDUC: An Uncertainty-driven Approach for Learning-based Robust Control","date":"2024-05-04","arxiv_id":"2405.02598","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-model-based-multi-agent-personalized-short","title":"A Model-based Multi-Agent Personalized Short-Video Recommender System","date":"2024-05-03","arxiv_id":"2405.01847","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-optimal-deterministic-policies-with","title":"Learning Optimal Deterministic Policies with Stochastic Policy Gradients","date":"2024-05-03","arxiv_id":"2405.02235","repositories_listed":0,"syntology":null}],"record_sha256":"f37fa78ab8e0654cf3c97b198c7037d0e3344efe238b16e4c5c73ad07ef2365a","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}