{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/59","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":59,"pages_in_order":135,"rows_per_page":100,"rows":[5801,5900],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/58","next":"/task/reinforcement-learning-2/papers/60","papers":[{"url":null,"slug":"an-information-theoretic-approach-to-12","title":"An Information Theoretic Approach to Interaction-Grounded Learning","date":"2024-01-10","arxiv_id":"2401.05015","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-decentralized-cooperative-multi-agent","title":"Fully Decentralized Cooperative Multi-Agent Reinforcement Learning: A Survey","date":"2024-01-10","arxiv_id":"2401.04934","repositories_listed":0,"syntology":null},{"url":null,"slug":"innate-values-driven-reinforcement-learning","title":"Innate-Values-driven Reinforcement Learning based Cooperative Multi-Agent Cognitive Modeling","date":"2024-01-10","arxiv_id":"2401.05572","repositories_listed":0,"syntology":null},{"url":null,"slug":"react-reinforcement-learning-for-controller","title":"ReACT: Reinforcement Learning for Controller Parametrization using B-Spline Geometries","date":"2024-01-10","arxiv_id":"2401.05251","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-optimizing-rag-for","title":"Reinforcement Learning for Optimizing RAG for Domain Chatbots","date":"2024-01-10","arxiv_id":"2401.06800","repositories_listed":0,"syntology":null},{"url":null,"slug":"taming-data-hungry-reinforcement-learning","title":"Taming \"data-hungry\" reinforcement learning? Stability in continuous state-action spaces","date":"2024-01-10","arxiv_id":"2401.05233","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-safe-load-balancing-based-on-control","title":"Towards Safe Load Balancing based on Control Barrier Functions and Deep Reinforcement Learning","date":"2024-01-10","arxiv_id":"2401.05525","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-spiking-actor-network-with-intra-layer","title":"Fully Spiking Actor Network with Intra-layer Connections for Reinforcement Learning","date":"2024-01-09","arxiv_id":"2401.05444","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-minimaximalist-approach-to-reinforcement","title":"A Minimaximalist Approach to Reinforcement Learning from Human Feedback","date":"2024-01-08","arxiv_id":"2401.04056","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-tensor-network-implementation-of-multi","title":"A Tensor Network Implementation of Multi Agent Reinforcement Learning","date":"2024-01-08","arxiv_id":"2401.03896","repositories_listed":0,"syntology":null},{"url":null,"slug":"guiding-drones-by-information-gain","title":"Guiding drones by information gain","date":"2024-01-08","arxiv_id":"2401.03947","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-with-sub","title":"Inverse Reinforcement Learning with Sub-optimal Experts","date":"2024-01-08","arxiv_id":"2401.03857","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-term-safe-reinforcement-learning-with","title":"Long-term Safe Reinforcement Learning with Binary Feedback","date":"2024-01-08","arxiv_id":"2401.03786","repositories_listed":0,"syntology":null},{"url":null,"slug":"clustercomm-discrete-communication-in","title":"ClusterComm: Discrete Communication in Decentralized MARL using Internal Representation Clustering","date":"2024-01-07","arxiv_id":"2401.03504","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-powered-code-vulnerability-repair-with","title":"LLM-Powered Code Vulnerability Repair with Reinforcement Learning and Semantic Reward","date":"2024-01-07","arxiv_id":"2401.03374","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-learning-via-dqn-for-log","title":"Semi-supervised learning via DQN for log anomaly detection","date":"2024-01-06","arxiv_id":"2401.03151","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-local-path","title":"Deep Reinforcement Learning for Local Path Following of an Autonomous Formula SAE Vehicle","date":"2024-01-05","arxiv_id":"2401.02903","repositories_listed":0,"syntology":null},{"url":null,"slug":"synergistic-formulaic-alpha-generation-for","title":"Synergistic Formulaic Alpha Generation for Quantitative Trading based on Reinforcement Learning","date":"2024-01-05","arxiv_id":"2401.02710","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-analyzing-generalization-in-deep","title":"A Survey Analyzing Generalization in Deep Reinforcement Learning","date":"2024-01-04","arxiv_id":"2401.02349","repositories_listed":0,"syntology":null},{"url":null,"slug":"glide-rl-grounded-language-instruction","title":"GLIDE-RL: Grounded Language Instruction through DEmonstration in RL","date":"2024-01-03","arxiv_id":"2401.02991","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-sar-view-angle","title":"Reinforcement Learning for SAR View Angle Inversion with Differentiable SAR Renderer","date":"2024-01-02","arxiv_id":"2401.01165","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-multi-task","title":"Meta Reinforcement Learning for Multi-Task Offloading in Vehicular Edge Computing","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-dynamic-pricing-policy-for","title":"Personalized Dynamic Pricing Policy for Electric Vehicles: Reinforcement learning approach","date":"2024-01-01","arxiv_id":"2401.00661","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularized-parameter-uncertainty-for","title":"Regularized Parameter Uncertainty for Improving Generalization in Reinforcement Learning","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"training-diffusion-models-towards-diverse","title":"Training Diffusion Models Towards Diverse Image Generation with Reinforcement Learning","date":"2024-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-learning-based-agent-modeling-for","title":"Contrastive learning-based agent modeling for deep reinforcement learning","date":"2023-12-30","arxiv_id":"2401.00132","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-two-phase-offline-deep","title":"Two-Step Offline Preference-Based Reinforcement Learning with Constrained Actions","date":"2023-12-30","arxiv_id":"2401.00330","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-space-exploration-of-approximate","title":"Design Space Exploration of Approximate Computing Techniques with a Reinforcement Learning Approach","date":"2023-12-29","arxiv_id":"2312.17525","repositories_listed":0,"syntology":null},{"url":null,"slug":"resilient-constrained-reinforcement-learning","title":"Resilient Constrained Reinforcement Learning","date":"2023-12-28","arxiv_id":"2312.17194","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-logo-deep-reinforcement-learning","title":"RL-LOGO: Deep Reinforcement Learning Localization for Logo Recognition","date":"2023-12-28","arxiv_id":"2312.16792","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlplanner-reinforcement-learning-based","title":"RLPlanner: Reinforcement Learning based Floorplanning for Chiplets with Fast Thermal Analysis","date":"2023-12-28","arxiv_id":"2312.16895","repositories_listed":0,"syntology":null},{"url":null,"slug":"foundations-of-reinforcement-learning-and","title":"Foundations of Reinforcement Learning and Interactive Decision Making","date":"2023-12-27","arxiv_id":"2312.16730","repositories_listed":0,"syntology":null},{"url":null,"slug":"general-method-for-solving-four-types-of-sat","title":"General Method for Solving Four Types of SAT Problems","date":"2023-12-27","arxiv_id":"2312.16423","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-closed-loop-multi-perspective-visual","title":"A Closed-Loop Multi-perspective Visual Servoing Approach with Reinforcement Learning","date":"2023-12-25","arxiv_id":"2312.15809","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-target-detection-algorithm-in-traffic","title":"A Target Detection Algorithm in Traffic Scenes Based on Deep Reinforcement Learning","date":"2023-12-25","arxiv_id":"2312.15606","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-quantitative","title":"Deep Reinforcement Learning for Quantitative Trading","date":"2023-12-25","arxiv_id":"2312.15730","repositories_listed":0,"syntology":null},{"url":null,"slug":"swap-based-deep-reinforcement-learning-for","title":"Swap-based Deep Reinforcement Learning for Facility Location Problems in Networks","date":"2023-12-25","arxiv_id":"2312.15658","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-based-modelling-for-continuously","title":"Agent based modelling for continuously varying supply chains","date":"2023-12-24","arxiv_id":"2312.15502","repositories_listed":0,"syntology":null},{"url":null,"slug":"discrete-time-mean-variance-strategy-based-on","title":"Discrete-Time Mean-Variance Strategy Based on Reinforcement Learning","date":"2023-12-24","arxiv_id":"2312.15385","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-based","title":"Distributional Reinforcement Learning-based Energy Arbitrage Strategies in Imbalance Settlement Mechanism","date":"2023-12-23","arxiv_id":"2401.00015","repositories_listed":0,"syntology":null},{"url":null,"slug":"gradient-shaping-for-multi-constraint-safe","title":"Gradient Shaping for Multi-Constraint Safe Reinforcement Learning","date":"2023-12-23","arxiv_id":"2312.15127","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-ai-collaboration-in-real-world-complex","title":"Human-AI Collaboration in Real-World Complex Environment with Reinforcement Learning","date":"2023-12-23","arxiv_id":"2312.15160","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-is-all-you-need-training-strong","title":"Scaling Is All You Need: Autonomous Driving with JAX-Accelerated Reinforcement Learning","date":"2023-12-23","arxiv_id":"2312.15122","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-note-on-stability-in-asynchronous","title":"A Note on Stability in Asynchronous Stochastic Approximation without Communication Delays","date":"2023-12-22","arxiv_id":"2312.15091","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-reinforcement-learning-from-human","title":"A Survey of Reinforcement Learning from Human Feedback","date":"2023-12-22","arxiv_id":"2312.14925","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-programming-based-approximate-optimal","title":"Dynamic Programming-based Approximate Optimal Control for Model-Based Reinforcement Learning","date":"2023-12-22","arxiv_id":"2312.14463","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-q-learning-linear-regret-speedup","title":"Federated Q-Learning: Linear Regret Speedup with Low Communication Cost","date":"2023-12-22","arxiv_id":"2312.15023","repositories_listed":0,"syntology":null},{"url":null,"slug":"rebel-a-regularization-based-solution-for","title":"REBEL: Reward Regularization-Based Approach for Robotic Reinforcement Learning from Human Feedback","date":"2023-12-22","arxiv_id":"2312.14436","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-1","title":"Safe Reinforcement Learning with Instantaneous Constraints: The Role of Aggressive Exploration","date":"2023-12-22","arxiv_id":"2312.14470","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-based-multiple","title":"A Reinforcement-Learning-Based Multiple-Column Selection Strategy for Column Generation","date":"2023-12-21","arxiv_id":"2312.14213","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-curriculum-learning-with-gradient","title":"Automatic Curriculum Learning with Gradient Reward Signals","date":"2023-12-21","arxiv_id":"2312.13565","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-multi-agent-preference-based","title":"Incorporating Human Flexibility through Reward Preferences in Human-AI Teaming","date":"2023-12-21","arxiv_id":"2312.14292","repositories_listed":0,"syntology":null},{"url":null,"slug":"cva-hedging-by-risk-averse-stochastic-horizon","title":"CVA Hedging by Risk-Averse Stochastic-Horizon Reinforcement Learning","date":"2023-12-21","arxiv_id":"2312.14044","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-coordination-in-minority-game-a","title":"Optimal coordination of resources: A solution from reinforcement learning","date":"2023-12-20","arxiv_id":"2312.14970","repositories_listed":0,"syntology":null},{"url":null,"slug":"pgn-a-perturbation-generation-network-against","title":"PGN: A perturbation generation network against deep reinforcement learning","date":"2023-12-20","arxiv_id":"2312.12904","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-mean-field-load-balancing-in-large","title":"Sparse Mean Field Load Balancing in Large Localized Queueing Systems","date":"2023-12-20","arxiv_id":"2312.12973","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-machines-that-trust-ai-agents-learn","title":"Towards Machines that Trust: AI Agents Learn to Trust in the Trust Game","date":"2023-12-20","arxiv_id":"2312.12868","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-learning-for-cooperation-in-multi","title":"Curriculum Learning for Cooperation in Multi-Agent Reinforcement Learning","date":"2023-12-19","arxiv_id":"2312.11768","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-network-approximation-for-pessimistic","title":"Neural Network Approximation for Pessimistic Offline Reinforcement Learning","date":"2023-12-19","arxiv_id":"2312.11863","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-search-and-coverage-using-point-cloud","title":"Active search and coverage using point-cloud reinforcement learning","date":"2023-12-18","arxiv_id":"2312.11410","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-reinforcement-learning-for","title":"Contextual Reinforcement Learning for Offshore Wind Farm Bidding","date":"2023-12-18","arxiv_id":"2312.10884","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-the-swing-up-and-balance-task-for-the","title":"Solving the swing-up and balance task for the Acrobot and Pendubot with SAC","date":"2023-12-18","arxiv_id":"2312.11311","repositories_listed":0,"syntology":null},{"url":null,"slug":"advancing-ran-slicing-with-offline","title":"Advancing RAN Slicing with Offline Reinforcement Learning","date":"2023-12-16","arxiv_id":"2312.10547","repositories_listed":0,"syntology":null},{"url":null,"slug":"fractional-deep-reinforcement-learning-for","title":"Fractional Deep Reinforcement Learning for Age-Minimal Mobile Edge Computing","date":"2023-12-16","arxiv_id":"2312.10418","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-communicative-multi-agent","title":"Robust Communicative Multi-Agent Reinforcement Learning with Active Defense","date":"2023-12-16","arxiv_id":"2312.11545","repositories_listed":0,"syntology":null},{"url":null,"slug":"assume-guarantee-reinforcement-learning","title":"Assume-Guarantee Reinforcement Learning","date":"2023-12-15","arxiv_id":"2312.09938","repositories_listed":0,"syntology":null},{"url":"/paper/graphrare-reinforcement-learning-enhanced","slug":"graphrare-reinforcement-learning-enhanced","title":"GraphRARE: Reinforcement Learning Enhanced Graph Neural Network with Relative Entropy","date":"2023-12-15","arxiv_id":"2312.09708","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-a-1","title":"Multi-agent Reinforcement Learning: A Comprehensive Survey","date":"2023-12-15","arxiv_id":"2312.10256","repositories_listed":0,"syntology":null},{"url":null,"slug":"pareto-envelope-augmented-with-reinforcement","title":"Multi-Objective Reinforcement Learning-based Approach for Pressurized Water Reactor Optimization","date":"2023-12-15","arxiv_id":"2312.10194","repositories_listed":0,"syntology":null},{"url":null,"slug":"situation-dependent-causal-influence-based","title":"Situation-Dependent Causal Influence-Based Cooperative Multi-agent Reinforcement Learning","date":"2023-12-15","arxiv_id":"2312.09539","repositories_listed":0,"syntology":null},{"url":null,"slug":"small-dataset-big-gains-enhancing","title":"Small Dataset, Big Gains: Enhancing Reinforcement Learning by Offline Pre-Training with Model Based Augmentation","date":"2023-12-15","arxiv_id":"2312.09844","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-computationally-efficient-inverse","title":"Toward Computationally Efficient Inverse Reinforcement Learning via Reward Shaping","date":"2023-12-15","arxiv_id":"2312.09983","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-parameter-sharing-for-multi-agent","title":"Adaptive parameter sharing for multi-agent reinforcement learning","date":"2023-12-14","arxiv_id":"2312.09009","repositories_listed":0,"syntology":null},{"url":null,"slug":"auto-mc-reward-automated-dense-reward-design","title":"Auto MC-Reward: Automated Dense Reward Design with Large Language Models for Minecraft","date":"2023-12-14","arxiv_id":"2312.09238","repositories_listed":0,"syntology":null},{"url":null,"slug":"improve-robustness-of-reinforcement-learning","title":"Improve Robustness of Reinforcement Learning against Observation Perturbations via $l_\\infty$ Lipschitz Policy Networks","date":"2023-12-14","arxiv_id":"2312.08751","repositories_listed":0,"syntology":null},{"url":null,"slug":"ion-profiler-intelligent-online-multi","title":"iOn-Profiler: intelligent Online multi-objective VNF Profiling with Reinforcement Learning","date":"2023-12-14","arxiv_id":"2312.09355","repositories_listed":0,"syntology":null},{"url":null,"slug":"lift-unsupervised-reinforcement-learning-with","title":"LiFT: Unsupervised Reinforcement Learning with Foundation Models as Teachers","date":"2023-12-14","arxiv_id":"2312.08958","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-path-recourse","title":"Personalized Path Recourse for Reinforcement Learning Agents","date":"2023-12-14","arxiv_id":"2312.08724","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-language-models-as-a-source-of-rewards","title":"Vision-Language Models as a Source of Rewards","date":"2023-12-14","arxiv_id":"2312.09187","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-invitation-to-deep-reinforcement-learning","title":"An Invitation to Deep Reinforcement Learning","date":"2023-12-13","arxiv_id":"2312.08365","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-robotic-navigation-an-evaluation-of","title":"Enhancing Robotic Navigation: An Evaluation of Single and Multi-Objective Reinforcement Learning Strategies","date":"2023-12-13","arxiv_id":"2312.07953","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-exploration-in-reinforcement-learning","title":"Safe Exploration in Reinforcement Learning: Training Backup Control Barrier Functions with Zero Training Time Safety Violations","date":"2023-12-13","arxiv_id":"2312.07828","repositories_listed":0,"syntology":null},{"url":null,"slug":"noise-distribution-decomposition-based-multi","title":"Noise Distribution Decomposition based Multi-Agent Distributional Reinforcement Learning","date":"2023-12-12","arxiv_id":"2312.07025","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-network-intrusion-detection-via","title":"Real-time Network Intrusion Detection via Decision Transformers","date":"2023-12-12","arxiv_id":"2312.07696","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-a-reinforcement-learning-based-system","title":"Toward a Reinforcement-Learning-Based System for Adjusting Medication to Minimize Speech Disfluency","date":"2023-12-12","arxiv_id":"2312.11509","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-decentralized-cooperative-platoon","title":"Scalable Decentralized Cooperative Platoon using Multi-Agent Deep Reinforcement Learning","date":"2023-12-11","arxiv_id":"2312.06858","repositories_listed":0,"syntology":null},{"url":null,"slug":"spreeze-high-throughput-parallel","title":"Spreeze: High-Throughput Parallel Reinforcement Learning Framework","date":"2023-12-11","arxiv_id":"2312.06126","repositories_listed":0,"syntology":null},{"url":null,"slug":"dcir-dynamic-consistency-intrinsic-reward-for","title":"DCIR: Dynamic Consistency Intrinsic Reward for Multi-Agent Reinforcement Learning","date":"2023-12-10","arxiv_id":"2312.05783","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-conditioned-semantic-search-based","title":"Language-Conditioned Semantic Search-Based Policy for Robotic Manipulation Tasks","date":"2023-12-10","arxiv_id":"2312.05925","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-reinforcement-learning-and-large","title":"PerfRL: A Small Language Model Framework for Efficient Code Optimization","date":"2023-12-09","arxiv_id":"2312.05657","repositories_listed":0,"syntology":null},{"url":null,"slug":"position-control-of-an-acoustic-cavitation","title":"Position control of an acoustic cavitation bubble by reinforcement learning","date":"2023-12-09","arxiv_id":"2312.05674","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-preserving-multi-agent-reinforcement","title":"Privacy Preserving Multi-Agent Reinforcement Learning in Supply Chains","date":"2023-12-09","arxiv_id":"2312.05686","repositories_listed":0,"syntology":null},{"url":null,"slug":"canaries-and-whistles-resilient-drone","title":"Canaries and Whistles: Resilient Drone Communication Networks with (or without) Deep Reinforcement Learning","date":"2023-12-08","arxiv_id":"2312.04940","repositories_listed":0,"syntology":null},{"url":null,"slug":"darlei-deep-accelerated-reinforcement","title":"DARLEI: Deep Accelerated Reinforcement Learning with Evolutionary Intelligence","date":"2023-12-08","arxiv_id":"2312.05171","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-risk-in-reinforcement-learning-a","title":"Modeling Risk in Reinforcement Learning: A Literature Mapping","date":"2023-12-08","arxiv_id":"2312.05231","repositories_listed":0,"syntology":null},{"url":null,"slug":"pruning-convolutional-filters-via","title":"Pruning Convolutional Filters via Reinforcement Learning with Entropy Minimization","date":"2023-12-08","arxiv_id":"2312.04918","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-active-feature-acquisition-1","title":"Evaluation of Active Feature Acquisition Methods for Static Feature Settings","date":"2023-12-06","arxiv_id":"2312.03619","repositories_listed":0,"syntology":null},{"url":null,"slug":"macca-offline-multi-agent-reinforcement","title":"MACCA: Offline Multi-agent Reinforcement Learning with Causal Credit Assignment","date":"2023-12-06","arxiv_id":"2312.03644","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-q-learning-approach-to-the-continuous","title":"A Q-learning approach to the continuous control problem of robot inverted pendulum balancing","date":"2023-12-05","arxiv_id":"2312.02649","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-reinforcement-learning-for-networked","title":"Provable Reinforcement Learning for Networked Control Systems with Stochastic Packet Disordering","date":"2023-12-05","arxiv_id":"2312.02498","repositories_listed":0,"syntology":null}],"record_sha256":"a9aadbc1fc55cbde31f55108e83a9a718fe45ebbbf440872300650487f6a2eaa","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}