{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/68","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":68,"pages_in_order":152,"rows_per_page":100,"rows":[6701,6800],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/67","next":"/task/reinforcement-learning-1/papers/69","papers":[{"url":null,"slug":"a-simple-way-to-incorporate-novelty-detection","title":"Novelty Detection in Reinforcement Learning with World Models","date":"2023-10-12","arxiv_id":"2310.08731","repositories_listed":0,"syntology":null},{"url":null,"slug":"discerning-temporal-difference-learning","title":"Discerning Temporal Difference Learning","date":"2023-10-12","arxiv_id":"2310.08091","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-evaluation-for-human-feedback","title":"Off-Policy Evaluation for Human Feedback","date":"2023-10-11","arxiv_id":"2310.07123","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-rl-in-linearly-q-p-realizable-mdps-is","title":"Online RL in Linearly $q^π$-Realizable MDPs Is as Easy as in Linear MDPs If You Learn What to Ignore","date":"2023-10-11","arxiv_id":"2310.07811","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-knowledge-graph","title":"Reinforcement Learning-based Knowledge Graph Reasoning for Explainable Fact-checking","date":"2023-10-11","arxiv_id":"2310.07613","repositories_listed":0,"syntology":null},{"url":null,"slug":"bi-level-offline-policy-optimization-with","title":"Bi-Level Offline Policy Optimization with Limited Exploration","date":"2023-10-10","arxiv_id":"2310.06268","repositories_listed":0,"syntology":null},{"url":null,"slug":"f-policy-gradients-a-general-framework-for","title":"$f$-Policy Gradients: A General Framework for Goal Conditioned RL using $f$-Divergences","date":"2023-10-10","arxiv_id":"2310.06794","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-a-safety-embedded","title":"Reinforcement Learning in a Safety-Embedded MDP with Trajectory Optimization","date":"2023-10-10","arxiv_id":"2310.06903","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-semantic-non-markovian-simulation","title":"Scalable Semantic Non-Markovian Simulation Proxy for Reinforcement Learning","date":"2023-10-10","arxiv_id":"2310.06835","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectral-entry-wise-matrix-estimation-for-low","title":"Spectral Entry-wise Matrix Estimation for Low-Rank Reinforcement Learning","date":"2023-10-10","arxiv_id":"2310.06793","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-timestep-models-for-model-based","title":"Multi-timestep models for Model-based Reinforcement Learning","date":"2023-10-09","arxiv_id":"2310.05672","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-double-descent-in-reinforcement-learning","title":"On Double Descent in Reinforcement Learning with LSTD and Random Features","date":"2023-10-09","arxiv_id":"2310.05518","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-auxiliary-objectives-in-deep-rl","title":"Predictive auxiliary objectives in deep RL mimic learning in the brain","date":"2023-10-09","arxiv_id":"2310.06089","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-is-agnostic-reinforcement-learning","title":"When is Agnostic Reinforcement Learning Statistically Tractable?","date":"2023-10-09","arxiv_id":"2310.06113","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributional-reinforcement-learning-with-7","title":"Distributional Reinforcement Learning with Online Risk-awareness Adaption","date":"2023-10-08","arxiv_id":"2310.05179","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-generalizable-agents-via-saliency","title":"Learning Generalizable Agents via Saliency-Guided Features Decorrelation","date":"2023-10-08","arxiv_id":"2310.05086","repositories_listed":0,"syntology":null},{"url":null,"slug":"lifelong-learning-for-fog-load-balancing-a","title":"Lifelong Learning for Fog Load Balancing: A Transfer Learning Approach","date":"2023-10-08","arxiv_id":"2310.05187","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-sequential-decision-making-in","title":"Optimal Sequential Decision-Making in Geosteering: A Reinforcement Learning Approach","date":"2023-10-07","arxiv_id":"2310.04772","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-ui-instruction-grounding-towards-a","title":"Reinforced UI Instruction Grounding: Towards a Generic UI Task Automation API","date":"2023-10-07","arxiv_id":"2310.04716","repositories_listed":0,"syntology":null},{"url":null,"slug":"sera-sample-efficient-reward-augmentation-in","title":"Improving Offline-to-Online Reinforcement Learning with Q Conditioned State Entropy Exploration","date":"2023-10-07","arxiv_id":"2310.19805","repositories_listed":0,"syntology":null},{"url":null,"slug":"adarec-adaptive-sequential-recommendation-for","title":"AURO: Reinforcement Learning for Adaptive User Retention Optimization in Recommender Systems","date":"2023-10-06","arxiv_id":"2310.03984","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-auxiliary-tasks-for-learning","title":"Improving Reinforcement Learning Efficiency with Auxiliary Tasks in Non-Visual Environments: A Comparison","date":"2023-10-06","arxiv_id":"2310.04241","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-confirming-transformer-for-locally","title":"Self-Confirming Transformer for Belief-Conditioned Adaptation in Offline Multi-Agent Reinforcement Learning","date":"2023-10-06","arxiv_id":"2310.04579","repositories_listed":0,"syntology":null},{"url":null,"slug":"constraint-conditioned-policy-optimization","title":"Constraint-Conditioned Policy Optimization for Versatile Safe Reinforcement Learning","date":"2023-10-05","arxiv_id":"2310.03718","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-the-level-sampling-process-impacts-zero","title":"How the level sampling process impacts zero-shot generalisation in deep reinforcement learning","date":"2023-10-05","arxiv_id":"2310.03494","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-control-of-district-cooling-energy","title":"Optimal Control of District Cooling Energy Plant with Reinforcement Learning and MPC","date":"2023-10-05","arxiv_id":"2310.03814","repositories_listed":0,"syntology":null},{"url":null,"slug":"resilient-legged-local-navigation-learning-to","title":"Resilient Legged Local Navigation: Learning to Traverse with Compromised Perception End-to-End","date":"2023-10-05","arxiv_id":"2310.03581","repositories_listed":0,"syntology":null},{"url":null,"slug":"rtdk-bo-high-dimensional-bayesian","title":"RTDK-BO: High Dimensional Bayesian Optimization with Reinforced Transformer Deep kernels","date":"2023-10-05","arxiv_id":"2310.03912","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-exploration-in-reinforcement-learning-a","title":"Safe Exploration in Reinforcement Learning: A Generalized Formulation and Algorithms","date":"2023-10-05","arxiv_id":"2310.03225","repositories_listed":0,"syntology":null},{"url":null,"slug":"decision-convformer-local-filtering-in","title":"Decision ConvFormer: Local Filtering in MetaFormer is Sufficient for Decision Making","date":"2023-10-04","arxiv_id":"2310.03022","repositories_listed":0,"syntology":null},{"url":null,"slug":"foundation-reinforcement-learning-towards","title":"Reinforcement Learning with Foundation Priors: Let the Embodied Agent Efficiently Learn on Its Own","date":"2023-10-04","arxiv_id":"2310.02635","repositories_listed":0,"syntology":null},{"url":null,"slug":"mathcal-b-coder-value-based-deep","title":"$\\mathcal{B}$-Coder: Value-Based Deep Reinforcement Learning for Program Synthesis","date":"2023-10-04","arxiv_id":"2310.03173","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-architecture-impact-on-identifying","title":"Neural architecture impact on identifying temporally extended Reinforcement Learning tasks","date":"2023-10-04","arxiv_id":"2310.03161","repositories_listed":0,"syntology":null},{"url":null,"slug":"proximal-policy-optimization-based-1","title":"Proximal Policy Optimization-Based Reinforcement Learning Approach for DC-DC Boost Converter Control: A Comparative Evaluation Against Traditional Control Techniques","date":"2023-10-04","arxiv_id":"2310.02945","repositories_listed":0,"syntology":null},{"url":null,"slug":"searching-for-high-value-molecules-using","title":"Searching for High-Value Molecules Using Reinforcement Learning and Transformers","date":"2023-10-04","arxiv_id":"2310.02902","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for-15","title":"A Deep Reinforcement Learning Approach for Interactive Search with Sentence-level Feedback","date":"2023-10-03","arxiv_id":"2310.03043","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligndiff-aligning-diverse-human-preferences","title":"AlignDiff: Aligning Diverse Human Preferences via Behavior-Customisable Diffusion Model","date":"2023-10-03","arxiv_id":"2310.02054","repositories_listed":0,"syntology":null},{"url":null,"slug":"blending-imitation-and-reinforcement-learning","title":"Blending Imitation and Reinforcement Learning for Robust Policy Improvement","date":"2023-10-03","arxiv_id":"2310.01737","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigating-uncertainty-in-esg-investing","title":"Navigating Uncertainty in ESG Investing","date":"2023-10-03","arxiv_id":"2310.02163","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-representation-complexity-of-model-based","title":"On Representation Complexity of Model-based and Model-free Reinforcement Learning","date":"2023-10-03","arxiv_id":"2310.01706","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-unified-framework-for-sequential","title":"Towards a Unified Framework for Sequential Decision Making","date":"2023-10-03","arxiv_id":"2310.02167","repositories_listed":0,"syntology":null},{"url":null,"slug":"pessimistic-nonlinear-least-squares-value","title":"Pessimistic Nonlinear Least-Squares Value Iteration for Offline Reinforcement Learning","date":"2023-10-02","arxiv_id":"2310.01380","repositories_listed":0,"syntology":null},{"url":null,"slug":"remedi-reinforcement-learning-driven-adaptive","title":"REMEDI: REinforcement learning-driven adaptive MEtabolism modeling of primary sclerosing cholangitis DIsease progression","date":"2023-10-02","arxiv_id":"2310.01426","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-bandits-model-to-deep-deterministic","title":"From Bandits Model to Deep Deterministic Policy Gradient, Reinforcement Learning with Contextual Information","date":"2023-10-01","arxiv_id":"2310.00642","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-control-of-an-inverted-pendulum-by-a","title":"Adaptive Control of an Inverted Pendulum by a Reinforcement Learning-based LQR Method","date":"2023-09-30","arxiv_id":"2310.04436","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-neural-style-transfer-with-deep","title":"Controlling Neural Style Transfer with Deep Reinforcement Learning","date":"2023-09-30","arxiv_id":"2310.00405","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-quantum-states-preparation-method-based-on","title":"A Quantum States Preparation Method Based on Difference-Driven Reinforcement Learning","date":"2023-09-29","arxiv_id":"2309.16972","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-driving-behavior-generation","title":"Adversarial Driving Behavior Generation Incorporating Human Risk Cognition for Autonomous Vehicle Evaluation","date":"2023-09-29","arxiv_id":"2310.00029","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-prompt-rewriting-for-personalized","title":"Learning to Rewrite Prompts for Personalized Text Generation","date":"2023-09-29","arxiv_id":"2310.00152","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-node-selection-in","title":"Reinforcement Learning for Node Selection in Branch-and-Bound","date":"2023-09-29","arxiv_id":"2310.00112","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficiency-separation-between-rl-methods","title":"Efficiency Separation between RL Methods: Model-Free, Model-Based and Goal-Conditioned","date":"2023-09-28","arxiv_id":"2309.16291","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-offline-reinforcement-learning-certify","title":"Robust Offline Reinforcement Learning -- Certify the Confidence Interval","date":"2023-09-28","arxiv_id":"2309.16631","repositories_listed":0,"syntology":null},{"url":null,"slug":"stackelberg-batch-policy-learning","title":"Stackelberg Batch Policy Learning","date":"2023-09-28","arxiv_id":"2309.16188","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-decision-transformer-for","title":"Uncertainty-Aware Decision Transformer for Stochastic Driving Environments","date":"2023-09-28","arxiv_id":"2309.16397","repositories_listed":0,"syntology":null},{"url":null,"slug":"raiju-reinforcement-learning-guided-post","title":"Raijū: Reinforcement Learning-Guided Post-Exploitation for Automating Security Assessment of Network Systems","date":"2023-09-27","arxiv_id":"2309.15518","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comparison-of-controller-architectures-and","title":"A comparison of controller architectures and learning mechanisms for arbitrary robot morphologies","date":"2023-09-25","arxiv_id":"2309.13908","repositories_listed":0,"syntology":null},{"url":null,"slug":"ode-based-recurrent-model-free-reinforcement","title":"ODE-based Recurrent Model-free Reinforcement Learning for POMDPs","date":"2023-09-25","arxiv_id":"2309.14078","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effectiveness-of-adversarial-samples","title":"On the Effectiveness of Adversarial Samples against Ensemble Learning-based Windows PE Malware Detectors","date":"2023-09-25","arxiv_id":"2309.13841","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-complexity-of-neural-policy-mirror","title":"Sample Complexity of Neural Policy Mirror Descent for Policy Optimization on Low-Dimensional Manifolds","date":"2023-09-25","arxiv_id":"2309.13915","repositories_listed":0,"syntology":null},{"url":null,"slug":"tracking-control-for-a-spherical-pendulum-via","title":"Tracking Control for a Spherical Pendulum via Curriculum Reinforcement Learning","date":"2023-09-25","arxiv_id":"2309.14096","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-offline-reinforcement-learning-for","title":"Boosting Offline Reinforcement Learning for Autonomous Driving with Hierarchical Latent Skills","date":"2023-09-24","arxiv_id":"2309.13614","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-reachability-estimation-for-safe","title":"Iterative Reachability Estimation for Safe Reinforcement Learning","date":"2023-09-24","arxiv_id":"2309.13528","repositories_listed":0,"syntology":null},{"url":null,"slug":"limits-of-actor-critic-algorithms-for","title":"Limits of Actor-Critic Algorithms for Decision Tree Policies Learning in IBMDPs","date":"2023-09-23","arxiv_id":"2309.13365","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-bandwidth-estimation-from-offline","title":"Offline to Online Learning for Real-Time Bandwidth Estimation","date":"2023-09-23","arxiv_id":"2309.13481","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-robust-header","title":"Reinforcement Learning for Robust Header Compression under Model Uncertainty","date":"2023-09-23","arxiv_id":"2309.13291","repositories_listed":0,"syntology":null},{"url":null,"slug":"h2o-an-improved-framework-for-hybrid-offline","title":"H2O+: An Improved Framework for Hybrid Offline-and-Online RL with Dynamics Gaps","date":"2023-09-22","arxiv_id":"2309.12716","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-offline-rl-from-internet-videos-via","title":"Robotic Offline RL from Internet Videos via Value-Function Pre-Training","date":"2023-09-22","arxiv_id":"2309.13041","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-reinforcement-learning-via","title":"Curriculum Reinforcement Learning via Morphology-Environment Co-Evolution","date":"2023-09-21","arxiv_id":"2309.12529","repositories_listed":0,"syntology":null},{"url":null,"slug":"delays-in-reinforcement-learning","title":"Delays in Reinforcement Learning","date":"2023-09-20","arxiv_id":"2309.11096","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-infinite","title":"Deep Reinforcement Learning for Infinite Horizon Mean Field Problems in Continuous Spaces","date":"2023-09-19","arxiv_id":"2309.10953","repositories_listed":0,"syntology":null},{"url":null,"slug":"physics-informed-machine-learning-for-data","title":"Physics-Informed Machine Learning for Data Anomaly Detection, Classification, Localization, and Mitigation: A Review, Challenges, and Path Forward","date":"2023-09-19","arxiv_id":"2309.10788","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-enabled-reinforcement-learning-for-time","title":"Graph-enabled Reinforcement Learning for Time Series Forecasting with Adaptive Intelligence","date":"2023-09-18","arxiv_id":"2309.10186","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-online-distillation-promoting-safe","title":"Guided Online Distillation: Promoting Safe Reinforcement Learning by Offline Demonstration","date":"2023-09-18","arxiv_id":"2309.09408","repositories_listed":0,"syntology":null},{"url":null,"slug":"mechanic-maker-2-0-reinforcement-learning-for","title":"Mechanic Maker 2.0: Reinforcement Learning for Evaluating Generated Rules","date":"2023-09-18","arxiv_id":"2309.09476","repositories_listed":0,"syntology":null},{"url":null,"slug":"privileged-to-predicted-towards-sensorimotor","title":"Privileged to Predicted: Towards Sensorimotor Reinforcement Learning for Urban Driving","date":"2023-09-18","arxiv_id":"2309.09756","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-reinforcement-learning-to-simplify","title":"Using Reinforcement Learning to Simplify Mealtime Insulin Dosing for People with Type 1 Diabetes: In-Silico Experiments","date":"2023-09-17","arxiv_id":"2309.09125","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-h-infinity-control-with-a-real","title":"Data-Driven H-infinity Control with a Real-Time and Efficient Reinforcement Learning Algorithm: An Application to Autonomous Mobility-on-Demand Systems","date":"2023-09-16","arxiv_id":"2309.08880","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-mildly-conservative-model-based","title":"DOMAIN: MilDly COnservative Model-BAsed OfflINe Reinforcement Learning","date":"2023-09-16","arxiv_id":"2309.08925","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-spiking-binary-neuron-detector-of-causal","title":"A Spiking Binary Neuron -- Detector of Causal Links","date":"2023-09-15","arxiv_id":"2309.08476","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantitative-and-qualitative-evaluation-of-1","title":"Autonomous and Human-Driven Vehicles Interacting in a Roundabout: A Quantitative and Qualitative Evaluation","date":"2023-09-15","arxiv_id":"2309.08254","repositories_listed":0,"syntology":null},{"url":null,"slug":"equivariant-data-augmentation-for","title":"Equivariant Data Augmentation for Generalization in Offline Reinforcement Learning","date":"2023-09-14","arxiv_id":"2309.07578","repositories_listed":0,"syntology":null},{"url":null,"slug":"physically-plausible-full-body-hand-object","title":"Physically Plausible Full-Body Hand-Object Interaction Synthesis","date":"2023-09-14","arxiv_id":"2309.07907","repositories_listed":0,"syntology":null},{"url":null,"slug":"proximal-bellman-mappings-for-reinforcement","title":"Proximal Bellman mappings for reinforcement learning and their application to robust adaptive filtering","date":"2023-09-14","arxiv_id":"2309.07548","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-real-world-quadrupedal-locomotion-benchmark","title":"A Real-World Quadrupedal Locomotion Benchmark for Offline Reinforcement Learning","date":"2023-09-13","arxiv_id":"2309.16718","repositories_listed":0,"syntology":null},{"url":null,"slug":"harnessing-deep-q-learning-for-enhanced","title":"Harnessing Deep Q-Learning for Enhanced Statistical Arbitrage in High-Frequency Trading: A Comprehensive Exploration","date":"2023-09-13","arxiv_id":"2311.10718","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-impact-of-action","title":"Investigating the Impact of Action Representations in Policy Gradient Algorithms","date":"2023-09-13","arxiv_id":"2309.06921","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-dual","title":"Safe Reinforcement Learning with Dual Robustness","date":"2023-09-13","arxiv_id":"2309.06835","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-aware-reinforcement-learning-through","title":"Risk-Aware Reinforcement Learning through Optimal Transport Theory","date":"2023-09-12","arxiv_id":"2309.06239","repositories_listed":0,"syntology":null},{"url":null,"slug":"update-monte-carlo-tree-search-umcts","title":"Improved Monte Carlo tree search formulation with multiple root nodes for discrete sizing optimization of truss structures","date":"2023-09-12","arxiv_id":"2309.06045","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-learning-in-low-rank-slate","title":"Representation Learning in Low-rank Slate-based Recommender Systems","date":"2023-09-10","arxiv_id":"2309.08622","repositories_listed":0,"syntology":null},{"url":null,"slug":"signal-temporal-logic-neural-predictive","title":"Signal Temporal Logic Neural Predictive Control","date":"2023-09-10","arxiv_id":"2309.05131","repositories_listed":0,"syntology":null},{"url":null,"slug":"advantage-actor-critic-with-reasoner","title":"Advantage Actor-Critic with Reasoner: Explaining the Agent's Behavior from an Exploratory Perspective","date":"2023-09-09","arxiv_id":"2309.04707","repositories_listed":0,"syntology":null},{"url":null,"slug":"verifiable-reinforcement-learning-systems-via","title":"Verifiable Reinforcement Learning Systems via Compositionality","date":"2023-09-09","arxiv_id":"2309.06420","repositories_listed":0,"syntology":null},{"url":null,"slug":"seeing-eye-quadruped-navigation-with-force","title":"Seeing-Eye Quadruped Navigation with Force Responsive Locomotion Control","date":"2023-09-08","arxiv_id":"2309.04370","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-state-representation-for-diminishing","title":"A State Representation for Diminishing Rewards","date":"2023-09-07","arxiv_id":"2309.03710","repositories_listed":0,"syntology":null},{"url":null,"slug":"reboot-reuse-data-for-bootstrapping-efficient","title":"REBOOT: Reuse Data for Bootstrapping Efficient Real-World Dexterous Manipulation","date":"2023-09-06","arxiv_id":"2309.03322","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-imitation-learning-algorithms","title":"A Survey of Imitation Learning: Algorithms, Recent Developments, and Challenges","date":"2023-09-05","arxiv_id":"2309.02473","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialog-action-aware-transformer-for-dialog","title":"Dialog Action-Aware Transformer for Dialog Policy Learning","date":"2023-09-05","arxiv_id":"2309.02240","repositories_listed":0,"syntology":null},{"url":null,"slug":"atms-algorithmic-trading-guided-market","title":"INTAGS: Interactive Agent-Guided Simulation","date":"2023-09-04","arxiv_id":"2309.01784","repositories_listed":0,"syntology":null},{"url":null,"slug":"hundreds-guide-millions-adaptive-offline","title":"Hundreds Guide Millions: Adaptive Offline Reinforcement Learning with Expert Guidance","date":"2023-09-04","arxiv_id":"2309.01448","repositories_listed":0,"syntology":null}],"record_sha256":"f718aa5c6b5188669bc754d404d929c81f871b4bce3dfc65f803cb469875f4aa","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}