{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/66","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":66,"pages_in_order":152,"rows_per_page":100,"rows":[6501,6600],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/65","next":"/task/reinforcement-learning-1/papers/67","papers":[{"url":null,"slug":"on-sample-efficient-offline-reinforcement-1","title":"On Sample-Efficient Offline Reinforcement Learning: Data Diversity, Posterior Sampling, and Beyond","date":"2024-01-06","arxiv_id":"2401.03301","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-uncertainty-aware-exploration","title":"A unified uncertainty-aware exploration: Combining epistemic and aleatory uncertainty","date":"2024-01-05","arxiv_id":"2401.02914","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-discounting-of-training-time-attacks","title":"Adaptive Discounting of Training Time Attacks","date":"2024-01-05","arxiv_id":"2401.02652","repositories_listed":0,"syntology":null},{"url":null,"slug":"synergistic-formulaic-alpha-generation-for","title":"Synergistic Formulaic Alpha Generation for Quantitative Trading based on Reinforcement Learning","date":"2024-01-05","arxiv_id":"2401.02710","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-survey-of-research-towards-ai","title":"A comprehensive survey of research towards AI-enabled unmanned aerial systems in pre-, active-, and post-wildfire management","date":"2024-01-04","arxiv_id":"2401.02456","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-an-adaptable-and-generalizable","title":"Towards an Adaptable and Generalizable Optimization Engine in Decision and Control: A Meta Reinforcement Learning Approach","date":"2024-01-04","arxiv_id":"2401.02508","repositories_listed":0,"syntology":null},{"url":null,"slug":"glide-rl-grounded-language-instruction","title":"GLIDE-RL: Grounded Language Instruction through DEmonstration in RL","date":"2024-01-03","arxiv_id":"2401.02991","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarially-trained-actor-critic-for-1","title":"Adversarially Trained Weighted Actor-Critic for Safe Offline Reinforcement Learning","date":"2024-01-01","arxiv_id":"2401.00629","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-dynamic-pricing-policy-for","title":"Personalized Dynamic Pricing Policy for Electric Vehicles: Reinforcement learning approach","date":"2024-01-01","arxiv_id":"2401.00661","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularized-parameter-uncertainty-for","title":"Regularized Parameter Uncertainty for Improving Generalization in Reinforcement Learning","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"training-diffusion-models-towards-diverse","title":"Training Diffusion Models Towards Diverse Image Generation with Reinforcement Learning","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tight-finite-time-bounds-of-two-time-scale","title":"Tight Finite Time Bounds of Two-Time-Scale Linear Stochastic Approximation with Markovian Noise","date":"2023-12-31","arxiv_id":"2401.00364","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-space-exploration-of-approximate","title":"Design Space Exploration of Approximate Computing Techniques with a Reinforcement Learning Approach","date":"2023-12-29","arxiv_id":"2312.17525","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-pid-controllers-ppo-with-neuralized","title":"Beyond PID Controllers: PPO with Neuralized PID Policy for Proton Beam Intensity Control in Mu2e","date":"2023-12-28","arxiv_id":"2312.17372","repositories_listed":0,"syntology":null},{"url":null,"slug":"resilient-constrained-reinforcement-learning","title":"Resilient Constrained Reinforcement Learning","date":"2023-12-28","arxiv_id":"2312.17194","repositories_listed":0,"syntology":null},{"url":null,"slug":"conversational-question-answering-with","title":"Conversational Question Answering with Reformulations over Knowledge Graph","date":"2023-12-27","arxiv_id":"2312.17269","repositories_listed":0,"syntology":null},{"url":null,"slug":"general-method-for-solving-four-types-of-sat","title":"General Method for Solving Four Types of SAT Problems","date":"2023-12-27","arxiv_id":"2312.16423","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-mpca-a-reinforcement-learning-based-multi","title":"RL-MPCA: A Reinforcement Learning Based Multi-Phase Computation Allocation Approach for Recommender Systems","date":"2023-12-27","arxiv_id":"2401.01369","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bayesian-framework-of-deep-reinforcement","title":"A Bayesian Framework of Deep Reinforcement Learning for Joint O-RAN/MEC Orchestration","date":"2023-12-26","arxiv_id":"2312.16142","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-online-policies-for-person-tracking","title":"Learning Online Policies for Person Tracking in Multi-View Environments","date":"2023-12-26","arxiv_id":"2312.15858","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-based-modelling-for-continuously","title":"Agent based modelling for continuously varying supply chains","date":"2023-12-24","arxiv_id":"2312.15502","repositories_listed":0,"syntology":null},{"url":null,"slug":"gradient-shaping-for-multi-constraint-safe","title":"Gradient Shaping for Multi-Constraint Safe Reinforcement Learning","date":"2023-12-23","arxiv_id":"2312.15127","repositories_listed":0,"syntology":null},{"url":null,"slug":"hardware-aware-dnn-compression-via-diverse","title":"Hardware-Aware DNN Compression via Diverse Pruning and Mixed-Precision Quantization","date":"2023-12-23","arxiv_id":"2312.15322","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-ai-collaboration-in-real-world-complex","title":"Human-AI Collaboration in Real-World Complex Environment with Reinforcement Learning","date":"2023-12-23","arxiv_id":"2312.15160","repositories_listed":0,"syntology":null},{"url":null,"slug":"mutual-information-as-intrinsic-reward-of","title":"Mutual Information as Intrinsic Reward of Reinforcement Learning Agents for On-demand Ride Pooling","date":"2023-12-23","arxiv_id":"2312.15195","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-safe-occupancy","title":"Reinforcement Learning for Safe Occupancy Strategies in Educational Spaces during an Epidemic","date":"2023-12-23","arxiv_id":"2312.15163","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-reinforcement-learning-from-human","title":"A Survey of Reinforcement Learning from Human Feedback","date":"2023-12-22","arxiv_id":"2312.14925","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiagent-copilot-approach-for-shared","title":"Multiagent Copilot Approach for Shared Autonomy between Human EEG and TD3 Deep Reinforcement Learning","date":"2023-12-22","arxiv_id":"2312.14458","repositories_listed":0,"syntology":null},{"url":null,"slug":"pangu-agent-a-fine-tunable-generalist-agent","title":"Pangu-Agent: A Fine-Tunable Generalist Agent with Structured Reasoning","date":"2023-12-22","arxiv_id":"2312.14878","repositories_listed":0,"syntology":null},{"url":null,"slug":"rebel-a-regularization-based-solution-for","title":"REBEL: Reward Regularization-Based Approach for Robotic Reinforcement Learning from Human Feedback","date":"2023-12-22","arxiv_id":"2312.14436","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximum-entropy-gflownets-with-soft-q","title":"Maximum entropy GFlowNets with soft Q-learning","date":"2023-12-21","arxiv_id":"2312.14331","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-probabilistic-ensembles-with","title":"Multi-Agent Probabilistic Ensembles with Trajectory Sampling for Connected Autonomous Vehicles","date":"2023-12-21","arxiv_id":"2312.13910","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-coordination-in-minority-game-a","title":"Optimal coordination of resources: A solution from reinforcement learning","date":"2023-12-20","arxiv_id":"2312.14970","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-machines-that-trust-ai-agents-learn","title":"Towards Machines that Trust: AI Agents Learn to Trust in the Trust Game","date":"2023-12-20","arxiv_id":"2312.12868","repositories_listed":0,"syntology":null},{"url":null,"slug":"cudc-a-curiosity-driven-unsupervised-data","title":"CUDC: A Curiosity-Driven Unsupervised Data Collection Method with Adaptive Temporal Distances for Offline Reinforcement Learning","date":"2023-12-19","arxiv_id":"2312.12191","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-merton-s-strategies-in-an-incomplete","title":"Data-Driven Merton's Strategies via Policy Randomization","date":"2023-12-19","arxiv_id":"2312.11797","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-network-approximation-for-pessimistic","title":"Neural Network Approximation for Pessimistic Offline Reinforcement Learning","date":"2023-12-19","arxiv_id":"2312.11863","repositories_listed":0,"syntology":null},{"url":null,"slug":"stable-relay-learning-optimization-approach","title":"Stable Relay Learning Optimization Approach for Fast Power System Production Cost Minimization Simulation","date":"2023-12-19","arxiv_id":"2312.11896","repositories_listed":0,"syntology":null},{"url":null,"slug":"taskflex-solver-for-multi-agent-pursuit-via","title":"A Dual Curriculum Learning Framework for Multi-UAV Pursuit-Evasion in Diverse Environments","date":"2023-12-19","arxiv_id":"2312.12255","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-search-and-coverage-using-point-cloud","title":"Active search and coverage using point-cloud reinforcement learning","date":"2023-12-18","arxiv_id":"2312.11410","repositories_listed":0,"syntology":null},{"url":null,"slug":"safeguarded-progress-in-reinforcement","title":"Safeguarded Progress in Reinforcement Learning: Safe Bayesian Exploration for Control Policy Synthesis","date":"2023-12-18","arxiv_id":"2312.11314","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-the-swing-up-and-balance-task-for-the","title":"Solving the swing-up and balance task for the Acrobot and Pendubot with SAC","date":"2023-12-18","arxiv_id":"2312.11311","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-ran-slicing-with-offline","title":"Advancing RAN Slicing with Offline Reinforcement Learning","date":"2023-12-16","arxiv_id":"2312.10547","repositories_listed":0,"syntology":null},{"url":null,"slug":"fractional-deep-reinforcement-learning-for","title":"Fractional Deep Reinforcement Learning for Age-Minimal Mobile Edge Computing","date":"2023-12-16","arxiv_id":"2312.10418","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-restless-multi-armed-bandits-with-long","title":"Online Restless Multi-Armed Bandits with Long-Term Fairness Constraints","date":"2023-12-16","arxiv_id":"2312.10303","repositories_listed":0,"syntology":null},{"url":null,"slug":"assume-guarantee-reinforcement-learning","title":"Assume-Guarantee Reinforcement Learning","date":"2023-12-15","arxiv_id":"2312.09938","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-computationally-efficient-inverse","title":"Toward Computationally Efficient Inverse Reinforcement Learning via Reward Shaping","date":"2023-12-15","arxiv_id":"2312.09983","repositories_listed":0,"syntology":null},{"url":null,"slug":"ion-profiler-intelligent-online-multi","title":"iOn-Profiler: intelligent Online multi-objective VNF Profiling with Reinforcement Learning","date":"2023-12-14","arxiv_id":"2312.09355","repositories_listed":0,"syntology":null},{"url":null,"slug":"recore-regularized-contrastive-representation","title":"ReCoRe: Regularized Contrastive Representation Learning of World Model","date":"2023-12-14","arxiv_id":"2312.09056","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-automatic-data-augmentation-for","title":"Towards Automatic Data Augmentation for Disordered Speech Recognition","date":"2023-12-14","arxiv_id":"2312.08641","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-invitation-to-deep-reinforcement-learning","title":"An Invitation to Deep Reinforcement Learning","date":"2023-12-13","arxiv_id":"2312.08365","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-exploration-in-reinforcement-learning","title":"Safe Exploration in Reinforcement Learning: Training Backup Control Barrier Functions with Zero Training Time Safety Violations","date":"2023-12-13","arxiv_id":"2312.07828","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-proximal-policy-optimization-with","title":"A dynamical clipping approach with task feedback for Proximal Policy Optimization","date":"2023-12-12","arxiv_id":"2312.07624","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-expected-return-accounting-for-policy","title":"Beyond Expected Return: Accounting for Policy Reproducibility when Evaluating Reinforcement Learning Algorithms","date":"2023-12-12","arxiv_id":"2312.07178","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-open-ended-embodied-agent-via","title":"Building Open-Ended Embodied Agent via Language-Policy Bidirectional Adaptation","date":"2023-12-12","arxiv_id":"2401.00006","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-distribution-decomposition-based-multi","title":"Noise Distribution Decomposition based Multi-Agent Distributional Reinforcement Learning","date":"2023-12-12","arxiv_id":"2312.07025","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-a-reinforcement-learning-based-system","title":"Toward a Reinforcement-Learning-Based System for Adjusting Medication to Minimize Speech Disfluency","date":"2023-12-12","arxiv_id":"2312.11509","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowgpt-black-box-knowledge-injection-for","title":"KnowGPT: Knowledge Graph based Prompting for Large Language Models","date":"2023-12-11","arxiv_id":"2312.06185","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-polynomial-representations-of","title":"Learning Polynomial Representations of Physical Objects with Application to Certifying Correct Packing Configurations","date":"2023-12-11","arxiv_id":"2312.06791","repositories_listed":0,"syntology":null},{"url":null,"slug":"partial-end-to-end-reinforcement-learning-for","title":"Partial End-to-end Reinforcement Learning for Robustness Against Modelling Error in Autonomous Racing","date":"2023-12-11","arxiv_id":"2312.06406","repositories_listed":0,"syntology":null},{"url":null,"slug":"spreeze-high-throughput-parallel","title":"Spreeze: High-Throughput Parallel Reinforcement Learning Framework","date":"2023-12-11","arxiv_id":"2312.06126","repositories_listed":0,"syntology":null},{"url":null,"slug":"modifying-rl-policies-with-imagined-actions","title":"Modifying RL Policies with Imagined Actions: How Predictable Policies Can Enable Users to Perform Novel Tasks","date":"2023-12-10","arxiv_id":"2312.05991","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-reinforcement-learning-and-large","title":"PerfRL: A Small Language Model Framework for Efficient Code Optimization","date":"2023-12-09","arxiv_id":"2312.05657","repositories_listed":0,"syntology":null},{"url":null,"slug":"guaranteed-trust-region-optimization-via-two","title":"Guaranteed Trust Region Optimization via Two-Phase KL Penalization","date":"2023-12-08","arxiv_id":"2312.05405","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-risk-in-reinforcement-learning-a","title":"Modeling Risk in Reinforcement Learning: A Literature Mapping","date":"2023-12-08","arxiv_id":"2312.05231","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-bionic-reflex","title":"Reinforcement Learning-Based Bionic Reflex Control for Anthropomorphic Robotic Grasping exploiting Domain Randomization","date":"2023-12-08","arxiv_id":"2312.05023","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-sample-in-cartesian-mri","title":"Learning to sample in Cartesian MRI","date":"2023-12-07","arxiv_id":"2312.04327","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-enhanced-self-learning-for-optimal","title":"Safety-Enhanced Self-Learning for Optimal Power Converter Control","date":"2023-12-07","arxiv_id":"2312.04158","repositories_listed":0,"syntology":null},{"url":null,"slug":"demand-response-for-residential-building","title":"Demand response for residential building heating: Effective Monte Carlo Tree Search control based on physics-informed neural networks","date":"2023-12-06","arxiv_id":"2312.03365","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffused-task-agnostic-milestone-planner-1","title":"Diffused Task-Agnostic Milestone Planner","date":"2023-12-06","arxiv_id":"2312.03395","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-active-feature-acquisition-1","title":"Evaluation of Active Feature Acquisition Methods for Static Feature Settings","date":"2023-12-06","arxiv_id":"2312.03619","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-role-of-the-action-space-in-robot","title":"On the Role of the Action Space in Robot Manipulation Learning and Sim-to-Real Transfer","date":"2023-12-06","arxiv_id":"2312.03673","repositories_listed":0,"syntology":null},{"url":null,"slug":"contact-energy-based-hindsight-experience","title":"Contact Energy Based Hindsight Experience Prioritization","date":"2023-12-05","arxiv_id":"2312.02677","repositories_listed":0,"syntology":null},{"url":null,"slug":"convergence-rates-for-stochastic-1","title":"Convergence Rates for Stochastic Approximation: Biased Noise with Unbounded Variance, and Applications","date":"2023-12-05","arxiv_id":"2312.02828","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitating-shortest-paths-in-simulation","title":"SPOC: Imitating Shortest Paths in Simulation Enables Effective Navigation and Manipulation in the Real World","date":"2023-12-05","arxiv_id":"2312.02976","repositories_listed":0,"syntology":null},{"url":null,"slug":"masp-scalable-gnn-based-planning-for-multi","title":"MASP: Scalable GNN-based Planning for Multi-Agent Navigation","date":"2023-12-05","arxiv_id":"2312.02522","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-based-cargo-uav-trajectory-planning-and","title":"RL-Based Cargo-UAV Trajectory Planning and Cell Association for Minimum Handoffs, Disconnectivity, and Energy Consumption","date":"2023-12-05","arxiv_id":"2312.02478","repositories_listed":0,"syntology":null},{"url":null,"slug":"score-aware-policy-gradient-methods-and","title":"Score-Aware Policy-Gradient Methods and Performance Guarantees using Local Lyapunov Conditions: Applications to Product-Form Stochastic Networks and Queueing Systems","date":"2023-12-05","arxiv_id":"2312.02804","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-operator-selection-utilising","title":"Adaptive operator selection utilising generalised experience","date":"2023-12-04","arxiv_id":"2401.05350","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-community","title":"Deep Reinforcement Learning for Community Battery Scheduling under Uncertainties of Load, PV Generation, and Energy Prices","date":"2023-12-04","arxiv_id":"2312.03008","repositories_listed":0,"syntology":null},{"url":null,"slug":"foundations-for-transfer-in-reinforcement","title":"Foundations for Transfer in Reinforcement Learning: A Taxonomy of Knowledge Modalities","date":"2023-12-04","arxiv_id":"2312.01939","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrated-drill-boom-hole-seeking-control","title":"Integrated Drill Boom Hole-Seeking Control via Reinforcement Learning","date":"2023-12-04","arxiv_id":"2312.01836","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-reinforcement-learning-agents-and","title":"Training Reinforcement Learning Agents and Humans With Difficulty-Conditioned Generators","date":"2023-12-04","arxiv_id":"2312.02309","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-critical-alternate-learning-based","title":"Self-Critical Alternate Learning based Semantic Broadcast Communication","date":"2023-12-03","arxiv_id":"2312.01423","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-multifidelity-sim-to-real-pipeline-for","title":"A Multifidelity Sim-to-Real Pipeline for Verifiable and Compositional Reinforcement Learning","date":"2023-12-02","arxiv_id":"2312.01249","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-temporal-credit-assignment-in","title":"A Survey of Temporal Credit Assignment in Deep Reinforcement Learning","date":"2023-12-02","arxiv_id":"2312.01072","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-off-policy-safe-reinforcement","title":"Efficient Off-Policy Safe Reinforcement Learning Using Trust Region Conditional Value at Risk","date":"2023-12-01","arxiv_id":"2312.00342","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-in-tensor","title":"Safe Reinforcement Learning in Tensor Reproducing Kernel Hilbert Space","date":"2023-12-01","arxiv_id":"2312.00727","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-deep-reinforcement-learning-2","title":"Data-efficient Deep Reinforcement Learning for Vehicle Trajectory Control","date":"2023-11-30","arxiv_id":"2311.18393","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-learning-based-optimal-false-data-injection","title":"Q-learning Based Optimal False Data Injection Attack on Probabilistic Boolean Control Networks","date":"2023-11-29","arxiv_id":"2311.17631","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-driving-telescopes-autonomous-scheduling","title":"Self-Driving Telescopes: Autonomous Scheduling of Astronomical Observation Campaigns with Offline Reinforcement Learning","date":"2023-11-29","arxiv_id":"2311.18094","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-step-reinforcement-learning-for","title":"Two-Step Reinforcement Learning for Multistage Strategy Card Game","date":"2023-11-29","arxiv_id":"2311.17305","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-in-a-simulated","title":"Safe Reinforcement Learning in a Simulated Robotic Arm","date":"2023-11-28","arxiv_id":"2312.09468","repositories_listed":0,"syntology":null},{"url":null,"slug":"2312-09436","title":"Temporal Transfer Learning for Traffic Optimization with Coarse-grained Advisory Autonomy","date":"2023-11-27","arxiv_id":"2312.09436","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-fully-data-driven-approach-for-realistic","title":"A Fully Data-Driven Approach for Realistic Traffic Signal Control Using Offline Reinforcement Learning","date":"2023-11-27","arxiv_id":"2311.15920","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-graph-neural-network-based-qubo-formulated","title":"A Graph Neural Network-Based QUBO-Formulated Hamiltonian-Inspired Loss Function for Combinatorial Optimization using Reinforcement Learning","date":"2023-11-27","arxiv_id":"2311.16277","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-observer-design-using-reinforcement","title":"Optimal Observer Design Using Reinforcement Learning and Quadratic Neural Networks","date":"2023-11-27","arxiv_id":"2311.16272","repositories_listed":0,"syntology":null},{"url":null,"slug":"replay-across-experiments-a-natural-extension","title":"Replay across Experiments: A Natural Extension of Off-Policy RL","date":"2023-11-27","arxiv_id":"2311.15951","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-nearly-optimal-and-low-switching-algorithm","title":"A Nearly Optimal and Low-Switching Algorithm for Reinforcement Learning with General Function Approximation","date":"2023-11-26","arxiv_id":"2311.15238","repositories_listed":0,"syntology":null},{"url":null,"slug":"projected-off-policy-q-learning-pop-ql-for","title":"Projected Off-Policy Q-Learning (POP-QL) for Stabilizing Offline Reinforcement Learning","date":"2023-11-25","arxiv_id":"2311.14885","repositories_listed":0,"syntology":null}],"record_sha256":"951f65724d15586577ad257438688aa7488c0b4514db2dc1abb9f65b40d21d4e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}