{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/74","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":74,"pages_in_order":152,"rows_per_page":100,"rows":[7301,7400],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/73","next":"/task/reinforcement-learning-1/papers/75","papers":[{"url":null,"slug":"deep-reinforcement-learning-for-multi-user","title":"Deep Reinforcement Learning for Multi-user Massive MIMO with Channel Aging","date":"2023-02-14","arxiv_id":"2302.06853","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-algorithms-applied-to-satellite","title":"Quantum algorithms applied to satellite mission planning for Earth observation","date":"2023-02-14","arxiv_id":"2302.07181","repositories_listed":0,"syntology":null},{"url":null,"slug":"to-risk-or-not-to-risk-learning-with-risk","title":"To Risk or Not to Risk: Learning with Risk Quantification for IoT Task Offloading in UAVs","date":"2023-02-14","arxiv_id":"2302.07399","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-lifetime-extended-energy-management","title":"A Lifetime Extended Energy Management Strategy for Fuel Cell Hybrid Electric Vehicles via Self-Learning Fuzzy Reinforcement Learning","date":"2023-02-13","arxiv_id":"2302.06236","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-modeling-long-term-user-engagement-from","title":"On Modeling Long-Term User Engagement from Stochastic Feedback","date":"2023-02-13","arxiv_id":"2302.06101","repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-agent-mixtures-and-the-geometry-of","title":"Universal Agent Mixtures and the Geometry of Intelligence","date":"2023-02-13","arxiv_id":"2302.06083","repositories_listed":0,"syntology":null},{"url":null,"slug":"maneuver-decision-making-for-autonomous-air","title":"Maneuver Decision-Making For Autonomous Air Combat Through Curriculum Learning And Reinforcement Learning With Sparse Rewards","date":"2023-02-12","arxiv_id":"2302.05838","repositories_listed":0,"syntology":null},{"url":null,"slug":"remix-regret-minimization-for-monotonic-value","title":"ReMIX: Regret Minimization for Monotonic Value Function Factorization in Multiagent Reinforcement Learning","date":"2023-02-11","arxiv_id":"2302.05593","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-causal-reinforcement-learning","title":"A Survey on Causal Reinforcement Learning","date":"2023-02-10","arxiv_id":"2302.05209","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-entropy-communication-in-multi-agent","title":"Low Entropy Communication in Multi-Agent Reinforcement Learning","date":"2023-02-10","arxiv_id":"2302.05055","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-minimax-optimality-of-model-based","title":"Towards Minimax Optimality of Model-based Robust Reinforcement Learning","date":"2023-02-10","arxiv_id":"2302.05372","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-into-pre-training-object","title":"An Investigation into Pre-Training Object-Centric Representations for Reinforcement Learning","date":"2023-02-09","arxiv_id":"2302.04419","repositories_listed":0,"syntology":null},{"url":null,"slug":"clare-conservative-model-based-reward","title":"CLARE: Conservative Model-Based Reward Learning for Offline Inverse Reinforcement Learning","date":"2023-02-09","arxiv_id":"2302.04782","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-quality-aware-mixed-precision","title":"Data Quality-aware Mixed-precision Quantization via Hybrid Reinforcement Learning","date":"2023-02-09","arxiv_id":"2302.04453","repositories_listed":0,"syntology":null},{"url":null,"slug":"equivariant-muzero","title":"Equivariant MuZero","date":"2023-02-09","arxiv_id":"2302.04798","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-near-optimal-algorithm-for-safe","title":"A Near-Optimal Algorithm for Safe Reinforcement Learning Under Instantaneous Hard Constraints","date":"2023-02-08","arxiv_id":"2302.04375","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-scale-independent-multi-objective","title":"A Scale-Independent Multi-Objective Reinforcement Learning with Convergence Analysis","date":"2023-02-08","arxiv_id":"2302.04179","repositories_listed":0,"syntology":null},{"url":null,"slug":"aisyn-ai-driven-reinforcement-learning-based","title":"AISYN: AI-driven Reinforcement Learning-Based Logic Synthesis Framework","date":"2023-02-08","arxiv_id":"2302.06415","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-planning-in-combinatorial-action","title":"Efficient Planning in Combinatorial Action Spaces with Applications to Cooperative Multi-Agent Reinforcement Learning","date":"2023-02-08","arxiv_id":"2302.04376","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-adversarial-reinforcement","title":"Near-Optimal Adversarial Reinforcement Learning with Switching Costs","date":"2023-02-08","arxiv_id":"2302.04374","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-aggregation-for-safety-critical","title":"Adaptive Aggregation for Safety-Critical Control","date":"2023-02-07","arxiv_id":"2302.03586","repositories_listed":0,"syntology":null},{"url":null,"slug":"eliciting-user-preferences-for-personalized","title":"Eliciting User Preferences for Personalized Multi-Objective Decision Making through Comparative Feedback","date":"2023-02-07","arxiv_id":"2302.03805","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-value-functions-for-efficient","title":"Ensemble Value Functions for Efficient Exploration in Multi-Agent Reinforcement Learning","date":"2023-02-07","arxiv_id":"2302.03439","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-minimax-optimal-risk-sensitive","title":"Near-Minimax-Optimal Risk-Sensitive Reinforcement Learning with CVaR","date":"2023-02-07","arxiv_id":"2302.03201","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-reinforcement-learning-with-uncertain","title":"Online Reinforcement Learning with Uncertain Episode Lengths","date":"2023-02-07","arxiv_id":"2302.03608","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-audio-recommendations-for-the-long","title":"Optimizing Audio Recommendations for the Long-Term: A Reinforcement Learning Perspective","date":"2023-02-07","arxiv_id":"2302.03561","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-skilled-population-curriculum-for","title":"Towards Skilled Population Curriculum for Multi-Agent Reinforcement Learning","date":"2023-02-07","arxiv_id":"2302.03429","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-for-process-design-with","title":"Transfer learning for process design with reinforcement learning","date":"2023-02-07","arxiv_id":"2302.03375","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-strong-baseline-for-batch-imitation","title":"A Strong Baseline for Batch Imitation Learning","date":"2023-02-06","arxiv_id":"2302.02788","repositories_listed":0,"syntology":null},{"url":null,"slug":"arena-web-a-web-based-development-and","title":"Arena-Web -- A Web-based Development and Benchmarking Platform for Autonomous Navigation Approaches","date":"2023-02-06","arxiv_id":"2302.02898","repositories_listed":0,"syntology":null},{"url":null,"slug":"ditto-offline-imitation-learning-with-world","title":"DITTO: Offline Imitation Learning with World Models","date":"2023-02-06","arxiv_id":"2302.03086","repositories_listed":0,"syntology":null},{"url":null,"slug":"holistic-deep-reinforcement-learning-based","title":"Holistic Deep-Reinforcement-Learning-based Training of Autonomous Navigation Systems","date":"2023-02-06","arxiv_id":"2302.02921","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-learning-in-markov-games-with-general","title":"Offline Learning in Markov Games with General Function Approximation","date":"2023-02-06","arxiv_id":"2302.02571","repositories_listed":0,"syntology":null},{"url":null,"slug":"rltp-reinforcement-learning-to-pace-for","title":"RLTP: Reinforcement Learning to Pace for Delayed Impression Modeling in Preloaded Ads","date":"2023-02-06","arxiv_id":"2302.02592","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-wise-safe-reinforcement-learning-a","title":"State-wise Safe Reinforcement Learning: A Survey","date":"2023-02-06","arxiv_id":"2302.03122","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-online-model-following-projection","title":"An Online Model-Following Projection Mechanism Using Reinforcement Learning","date":"2023-02-05","arxiv_id":"2302.02493","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-problems-and-modern-solutions-for-deep","title":"Open Problems and Modern Solutions for Deep Reinforcement Learning","date":"2023-02-05","arxiv_id":"2302.02298","repositories_listed":0,"syntology":null},{"url":null,"slug":"refined-value-based-offline-rl-under","title":"Offline Minimax Soft-Q-learning Under Realizability and Partial Coverage","date":"2023-02-05","arxiv_id":"2302.02392","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-traffic-light-1","title":"Deep Reinforcement Learning for Traffic Light Control in Intelligent Transportation Systems","date":"2023-02-04","arxiv_id":"2302.03669","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalization-of-deep-reinforcement-learning","title":"Generalization of Deep Reinforcement Learning for Jammer-Resilient Frequency and Power Allocation","date":"2023-02-04","arxiv_id":"2302.02250","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-learning-with-unsupervised-skill","title":"Developing Driving Strategies Efficiently: A Skill-Based Hierarchical Reinforcement Learning Approach","date":"2023-02-04","arxiv_id":"2302.02179","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-low-rank-mdps-with","title":"Reinforcement Learning in Low-Rank MDPs with Density Features","date":"2023-02-04","arxiv_id":"2302.02252","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-history-dependent","title":"Reinforcement Learning with History-Dependent Dynamic Contexts","date":"2023-02-04","arxiv_id":"2302.02061","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-cyber-system","title":"Deep Reinforcement Learning for Cyber System Defense under Dynamic Adversarial Uncertainties","date":"2023-02-03","arxiv_id":"2302.01595","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-online-error","title":"Deep Reinforcement Learning for Online Error Detection in Cyber-Physical Systems","date":"2023-02-03","arxiv_id":"2302.01567","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcing-user-retention-in-a-billion-scale","title":"Reinforcing User Retention in a Billion Scale Short Video Recommender System","date":"2023-02-03","arxiv_id":"2302.01724","repositories_listed":0,"syntology":null},{"url":"/paper/average-constrained-policy-optimization","slug":"average-constrained-policy-optimization","title":"ACPO: A Policy Optimization Algorithm for Average MDPs with Constraints","date":"2023-02-02","arxiv_id":"2302.00808","repositories_listed":0,"syntology":{"n":12,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/average-constrained-policy-optimization#ran","syntology_url":"https://syntology.ai/paper/2302.00808","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00808"}},"official":null}},{"url":null,"slug":"diversity-through-exclusion-dte-niche","title":"Diversity Through Exclusion (DTE): Niche Identification for Reinforcement Learning through Value-Decomposition","date":"2023-02-02","arxiv_id":"2302.01180","repositories_listed":0,"syntology":null},{"url":null,"slug":"lower-bounds-for-learning-in-revealing-pomdps","title":"Lower Bounds for Learning in Revealing POMDPs","date":"2023-02-02","arxiv_id":"2302.01333","repositories_listed":0,"syntology":null},{"url":null,"slug":"marlin-soft-actor-critic-based-reinforcement","title":"MARLIN: Soft Actor-Critic based Reinforcement Learning for Congestion Control in Real Networks","date":"2023-02-02","arxiv_id":"2302.01301","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-bounds-for-policy-based-average","title":"Performance Bounds for Policy-Based Average Reward Reinforcement Learning Algorithms","date":"2023-02-02","arxiv_id":"2302.01450","repositories_listed":0,"syntology":null},{"url":null,"slug":"reload-reinforcement-learning-with-optimistic","title":"ReLOAD: Reinforcement Learning with Optimistic Ascent-Descent for Last-Iterate Convergence in Constrained MDPs","date":"2023-02-02","arxiv_id":"2302.01275","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-physics-informed-neural-networks","title":"Bridging Physics-Informed Neural Networks with Reinforcement Learning: Hamilton-Jacobi-Bellman Proximal Policy Optimization (HJBPPO)","date":"2023-02-01","arxiv_id":"2302.00237","repositories_listed":0,"syntology":null},{"url":"/paper/collaborating-with-language-models-for","slug":"collaborating-with-language-models-for","title":"Collaborating with language models for embodied reasoning","date":"2023-02-01","arxiv_id":"2302.00763","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/collaborating-with-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2302.00763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00763"}},"official":null}},{"url":null,"slug":"combining-tree-search-generative-models-and","title":"Combining Deep Reinforcement Learning and Search with Generative Models for Game-Theoretic Opponent Modeling","date":"2023-02-01","arxiv_id":"2302.00797","repositories_listed":0,"syntology":null},{"url":"/paper/efficient-multi-task-reinforcement-learning","slug":"efficient-multi-task-reinforcement-learning","title":"QMP: Q-switch Mixture of Policies for Multi-Task Behavior Sharing","date":"2023-02-01","arxiv_id":"2302.00671","repositories_listed":0,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-multi-task-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2302.00671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00671"}},"official":null}},{"url":null,"slug":"multi-zone-hvac-control-with-model-based-deep","title":"Multi-zone HVAC Control with Model-Based Deep Reinforcement Learning","date":"2023-02-01","arxiv_id":"2302.00725","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-complexity-of-kernel-based-q-learning","title":"Sample Complexity of Kernel-Based Q-Learning","date":"2023-02-01","arxiv_id":"2302.00727","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-uncertainty-propagation-in-offline","title":"Selective Uncertainty Propagation in Offline RL","date":"2023-02-01","arxiv_id":"2302.00284","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-surrogate-assisted-evolutionary","title":"Enabling surrogate-assisted evolutionary reinforcement learning via policy embedding","date":"2023-01-31","arxiv_id":"2301.13374","repositories_listed":0,"syntology":null},{"url":null,"slug":"partitioning-distributed-compute-jobs-with","title":"Partitioning Distributed Compute Jobs with Reinforcement Learning and Graph Neural Networks","date":"2023-01-31","arxiv_id":"2301.13799","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-and-decision-making","title":"Towards interpretable quantum machine learning via single-photon quantum walks","date":"2023-01-31","arxiv_id":"2301.13669","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-grid-aware-dynamic-matching-using","title":"Scalable Grid-Aware Dynamic Matching using Deep Reinforcement Learning","date":"2023-01-31","arxiv_id":"2301.13796","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-laws-for-single-agent-reinforcement","title":"Scaling laws for single-agent reinforcement learning","date":"2023-01-31","arxiv_id":"2301.13442","repositories_listed":0,"syntology":null},{"url":null,"slug":"scheduling-inference-workloads-on-distributed","title":"Scheduling Inference Workloads on Distributed Edge Clusters with Reinforcement Learning","date":"2023-01-31","arxiv_id":"2301.13618","repositories_listed":0,"syntology":null},{"url":null,"slug":"skill-decision-transformer","title":"Skill Decision Transformer","date":"2023-01-31","arxiv_id":"2301.13573","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-programmatic-reinforcement","title":"Hierarchical Programmatic Reinforcement Learning via Learning to Compose Programs","date":"2023-01-30","arxiv_id":"2301.12950","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-regret-for-efficient-online","title":"Improved Regret for Efficient Online Reinforcement Learning with Linear Function Approximation","date":"2023-01-30","arxiv_id":"2301.13087","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-bounds-for-markov-decision-processes","title":"Regret Bounds for Markov Decision Processes with Recursive Optimized Certainty Equivalents","date":"2023-01-30","arxiv_id":"2301.12601","repositories_listed":0,"syntology":null},{"url":null,"slug":"singularity-aware-reinforcement-learning","title":"STEEL: Singularity-aware Reinforcement Learning","date":"2023-01-30","arxiv_id":"2301.13152","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferring-multiple-policies-to-hotstart","title":"Transferring Multiple Policies to Hotstart Reinforcement Learning in an Air Compressor Management Problem","date":"2023-01-30","arxiv_id":"2301.12820","repositories_listed":0,"syntology":null},{"url":null,"slug":"v2n-service-scaling-with-deep-reinforcement","title":"V2N Service Scaling with Deep Reinforcement Learning","date":"2023-01-30","arxiv_id":"2301.13324","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-framework-for-7","title":"A Deep Reinforcement Learning Framework for Optimizing Congestion Control in Data Centers","date":"2023-01-29","arxiv_id":"2301.12558","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-satellite-docking-via-adaptive","title":"Autonomous Satellite Docking via Adaptive Optimal Output Rregulation: A Reinforcement Learning Approach","date":"2023-01-29","arxiv_id":"2301.12489","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-deep-reinforcement-learning-3","title":"Sample Efficient Deep Reinforcement Learning via Local Planning","date":"2023-01-29","arxiv_id":"2301.12579","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-exponentially-fast-mixing-in-average","title":"Beyond Exponentially Fast Mixing in Average-Reward Reinforcement Learning via Multi-Level Monte Carlo Actor-Critic","date":"2023-01-28","arxiv_id":"2301.12083","repositories_listed":0,"syntology":null},{"url":null,"slug":"saformer-a-conditional-sequence-modeling","title":"SaFormer: A Conditional Sequence Modeling Approach to Offline Safe Reinforcement Learning","date":"2023-01-28","arxiv_id":"2301.12203","repositories_listed":0,"syntology":null},{"url":null,"slug":"steering-stein-information-directed","title":"STEERING: Stein Information Directed Exploration for Model-Based Reinforcement Learning","date":"2023-01-28","arxiv_id":"2301.12038","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-learning-rubik-s-cube-with-n-tuple","title":"Towards Learning Rubik's Cube with N-tuple-based Reinforcement Learning","date":"2023-01-28","arxiv_id":"2301.12167","repositories_listed":0,"syntology":null},{"url":null,"slug":"turbulence-control-in-plane-couette-flow","title":"Turbulence control in plane Couette flow using low-dimensional neural ODE-based models and deep reinforcement learning","date":"2023-01-28","arxiv_id":"2301.12098","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-memory-efficient-deep-reinforcement","title":"A Memory Efficient Deep Reinforcement Learning Approach For Snake Game Autonomous Agents","date":"2023-01-27","arxiv_id":"2301.11977","repositories_listed":0,"syntology":null},{"url":null,"slug":"behaviour-discriminator-a-simple-data","title":"Improving Behavioural Cloning with Positive Unlabeled Learning","date":"2023-01-27","arxiv_id":"2301.11734","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-deep-reinforcement-learning-for","title":"Exploring Deep Reinforcement Learning for Holistic Smart Building Control","date":"2023-01-27","arxiv_id":"2301.11510","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-munchausen-reinforcement-learning","title":"Generalized Munchausen Reinforcement Learning using Tsallis KL Divergence","date":"2023-01-27","arxiv_id":"2301.11476","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-human-road-crossing-decisions-as","title":"Modeling human road crossing decisions as reward maximization with visual perception limitations","date":"2023-01-27","arxiv_id":"2301.11737","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-from-diverse-human","title":"Reinforcement Learning from Diverse Human Preferences","date":"2023-01-27","arxiv_id":"2301.11774","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-trajectory-distributionally-robust","title":"Single-Trajectory Distributionally Robust Reinforcement Learning","date":"2023-01-27","arxiv_id":"2301.11721","repositories_listed":0,"syntology":null},{"url":null,"slug":"snerl-semantic-aware-neural-radiance-fields","title":"SNeRL: Semantic-aware Neural Radiance Fields for Reinforcement Learning","date":"2023-01-27","arxiv_id":"2301.11520","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-constrained-reinforcement-learning","title":"Solving Richly Constrained Reinforcement Learning through State Augmentation and Reward Penalties","date":"2023-01-27","arxiv_id":"2301.11592","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-regret-minimization-in-multi","title":"Communication-Efficient Collaborative Regret Minimization in Multi-Armed Bandits","date":"2023-01-26","arxiv_id":"2301.11442","repositories_listed":0,"syntology":null},{"url":null,"slug":"fedhql-federated-heterogeneous-q-learning","title":"FedHQL: Federated Heterogeneous Q-Learning","date":"2023-01-26","arxiv_id":"2301.11135","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-generate-all-feasible-actions","title":"Learning to Generate All Feasible Actions","date":"2023-01-26","arxiv_id":"2301.11461","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-offline-reinforcement-learning-1","title":"Model-based Offline Reinforcement Learning with Local Misspecification","date":"2023-01-26","arxiv_id":"2301.11426","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-global-convergence-of-risk-averse","title":"On the Global Convergence of Risk-Averse Policy Gradient Methods with Expected Conditional Risk Measures","date":"2023-01-26","arxiv_id":"2301.10932","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-optimization-with-robustness","title":"Certifiably Robust Reinforcement Learning through Model-Based Abstract Interpretation","date":"2023-01-26","arxiv_id":"2301.11374","repositories_listed":0,"syntology":null},{"url":null,"slug":"principled-reinforcement-learning-with-human","title":"Principled Reinforcement Learning with Human Feedback from Pairwise or $K$-wise Comparisons","date":"2023-01-26","arxiv_id":"2301.11270","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-neural-network-algorithm-for-linear","title":"A Deep Neural Network Algorithm for Linear-Quadratic Portfolio Optimization with MGARCH and Small Transaction Costs","date":"2023-01-25","arxiv_id":"2301.10869","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-deep-reinforcement-learning-based-2","title":"A Novel Deep Reinforcement Learning-based Approach for Enhancing Spectral Efficiency of IRS-assisted Wireless Systems","date":"2023-01-24","arxiv_id":"2302.14706","repositories_listed":0,"syntology":null},{"url":null,"slug":"asq-it-interactive-explanations-for","title":"ASQ-IT: Interactive Explanations for Reinforcement-Learning Agents","date":"2023-01-24","arxiv_id":"2301.09941","repositories_listed":0,"syntology":null},{"url":null,"slug":"autocost-evolving-intrinsic-cost-for-zero","title":"AutoCost: Evolving Intrinsic Cost for Zero-violation Reinforcement Learning","date":"2023-01-24","arxiv_id":"2301.10339","repositories_listed":0,"syntology":null}],"record_sha256":"d6fb56ceb69462dc1c808b89640f4400c429409d7a364058b797683f776c49df","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}