{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/65","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":65,"pages_in_order":135,"rows_per_page":100,"rows":[6401,6500],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/64","next":"/task/reinforcement-learning-2/papers/66","papers":[{"url":null,"slug":"reinforcement-learning-for-syntax-guided","title":"Reinforcement Learning and Data-Generation for Syntax-Guided Synthesis","date":"2023-07-13","arxiv_id":"2307.09564","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-complexity-of-non-stationary","title":"The complexity of non-stationary reinforcement learning","date":"2023-07-13","arxiv_id":"2307.06877","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-learning-for-autonomous-vehicle-s","title":"Machine Learning for Autonomous Vehicle's Trajectory Prediction: A comprehensive survey, Challenges, and Future Research Directions","date":"2023-07-12","arxiv_id":"2307.07527","repositories_listed":0,"syntology":null},{"url":null,"slug":"maneuver-decision-making-through-automatic","title":"Maneuver Decision-Making Through Automatic Curriculum Reinforcement Learning Without Handcrafted Reward functions","date":"2023-07-12","arxiv_id":"2307.06152","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompt-generate-train-pgt-a-framework-for-few","title":"Prompt Generate Train (PGT): Few-shot Domain Adaption of Retrieval Augmented Generation Models for Open Book Question-Answering","date":"2023-07-12","arxiv_id":"2307.05915","repositories_listed":0,"syntology":null},{"url":null,"slug":"transformers-in-reinforcement-learning-a","title":"Transformers in Reinforcement Learning: A Survey","date":"2023-07-12","arxiv_id":"2307.05979","repositories_listed":0,"syntology":null},{"url":"/paper/model-card-and-evaluations-for-claude-models","slug":"model-card-and-evaluations-for-claude-models","title":"Model Card and Evaluations for Claude Models","date":"2023-07-11","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multiobjective-hydropower-reservoir-operation","title":"Multiobjective Hydropower Reservoir Operation Optimization with Transformer-Based Deep Reinforcement Learning","date":"2023-07-11","arxiv_id":"2307.05643","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-distributed-multi-task-reinforcement","title":"Scaling Distributed Multi-task Reinforcement Learning with Experience Sharing","date":"2023-07-11","arxiv_id":"2307.05834","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-learning-as-computationally","title":"Continual Learning as Computationally Constrained Reinforcement Learning","date":"2023-07-10","arxiv_id":"2307.04345","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-policies-for-out-of-distribution","title":"Diffusion Policies for Out-of-Distribution Generalization in Offline Reinforcement Learning","date":"2023-07-10","arxiv_id":"2307.04726","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-interpretable-heuristics-for-walksat","title":"Learning Interpretable Heuristics for WalkSAT","date":"2023-07-10","arxiv_id":"2307.04608","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-and-mitigating-interference-in-1","title":"Measuring and Mitigating Interference in Reinforcement Learning","date":"2023-07-10","arxiv_id":"2307.04887","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-personalized-reinforcement-learning","title":"A Personalized Reinforcement Learning Summarization Service for Learning Structure from Unstructured Data","date":"2023-07-09","arxiv_id":"2307.05696","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-user-study-on-explainable-online","title":"A User Study on Explainable Online Reinforcement Learning for Adaptive Systems","date":"2023-07-09","arxiv_id":"2307.04098","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigating-the-edge-of-stability","title":"Investigating the Edge of Stability Phenomenon in Reinforcement Learning","date":"2023-07-09","arxiv_id":"2307.04210","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-trajectory-optimization-for-directional","title":"UAV Trajectory Optimization for Directional THz Links Using Deep Reinforcement Learning","date":"2023-07-08","arxiv_id":"2307.05535","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-conditioned-predictive-coding-as-an","title":"Goal-Conditioned Predictive Coding for Offline Reinforcement Learning","date":"2023-07-07","arxiv_id":"2307.03406","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-and-deep-reinforcement-learning","title":"Reinforcement and Deep Reinforcement Learning-based Solutions for Machine Maintenance Planning, Scheduling Policies, and Optimization","date":"2023-07-07","arxiv_id":"2307.03860","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-neuromorphic-architecture-for-reinforcement","title":"A Neuromorphic Architecture for Reinforcement Learning from Real-Valued Observations","date":"2023-07-06","arxiv_id":"2307.02947","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-federated-reinforcement-learning-for","title":"Meta Federated Reinforcement Learning for Distributed Resource Allocation","date":"2023-07-06","arxiv_id":"2307.02900","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-7","title":"Offline Reinforcement Learning with Imbalanced Datasets","date":"2023-07-06","arxiv_id":"2307.02752","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-iterated-cvar","title":"Provably Efficient Iterated CVaR Reinforcement Learning with Function Approximation and Human Feedback","date":"2023-07-06","arxiv_id":"2307.02842","repositories_listed":0,"syntology":null},{"url":null,"slug":"tgrl-an-algorithm-for-teacher-guided","title":"TGRL: An Algorithm for Teacher Guided Reinforcement Learning","date":"2023-07-06","arxiv_id":"2307.03186","repositories_listed":0,"syntology":null},{"url":null,"slug":"llql-logistic-likelihood-q-learning-for","title":"LLQL: Logistic Likelihood Q-Learning for Reinforcement Learning","date":"2023-07-05","arxiv_id":"2307.02345","repositories_listed":0,"syntology":null},{"url":null,"slug":"surge-routing-event-informed-multiagent","title":"Surge Routing: Event-informed Multiagent Reinforcement Learning for Autonomous Rideshare","date":"2023-07-05","arxiv_id":"2307.02637","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-conservatism-diffusion-policies-in","title":"Beyond Conservatism: Diffusion Policies in Offline Multi-agent Reinforcement Learning","date":"2023-07-04","arxiv_id":"2307.01472","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-reinforcement-learning-a-survey","title":"Causal Reinforcement Learning: A Survey","date":"2023-07-04","arxiv_id":"2307.01452","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergent-resource-exchange-and-tolerated","title":"Emergent Resource Exchange and Tolerated Theft Behavior using Multi-Agent Reinforcement Learning","date":"2023-07-04","arxiv_id":"2307.01862","repositories_listed":0,"syntology":null},{"url":null,"slug":"market-making-of-options-via-reinforcement","title":"Option Market Making via Reinforcement Learning","date":"2023-07-04","arxiv_id":"2307.01814","repositories_listed":0,"syntology":null},{"url":null,"slug":"over-the-counter-market-making-via","title":"Over-the-Counter Market Making via Reinforcement Learning","date":"2023-07-04","arxiv_id":"2307.01816","repositories_listed":0,"syntology":null},{"url":null,"slug":"achieving-stable-training-of-reinforcement","title":"Achieving Stable Training of Reinforcement Learning Agents in Bimodal Environments through Batch Learning","date":"2023-07-03","arxiv_id":"2307.00923","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-the-robustness-of-qmix-against","title":"Enhancing the Robustness of QMIX against State-adversarial Attacks","date":"2023-07-03","arxiv_id":"2307.00907","repositories_listed":0,"syntology":null},{"url":null,"slug":"theory-of-mind-as-intrinsic-motivation-for","title":"Theory of Mind as Intrinsic Motivation for Multi-Agent Reinforcement Learning","date":"2023-07-03","arxiv_id":"2307.01158","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-risk-sensitive-reinforcement-learning","title":"Is Risk-Sensitive Reinforcement Learning Properly Resolved?","date":"2023-07-02","arxiv_id":"2307.00547","repositories_listed":0,"syntology":null},{"url":null,"slug":"asset-correlation-based-deep-reinforcement","title":"Asset correlation based deep reinforcement learning for the portfolio selection","date":"2023-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-reinforcement-learning-and-human","title":"Comparing Reinforcement Learning and Human Learning using the Game of Hidden Rules","date":"2023-06-30","arxiv_id":"2306.17766","repositories_listed":0,"syntology":null},{"url":null,"slug":"design-of-induction-machines-using","title":"Design of Induction Machines using Reinforcement Learning","date":"2023-06-30","arxiv_id":"2306.17626","repositories_listed":0,"syntology":null},{"url":null,"slug":"l-ac-learning-latent-decision-aware-models","title":"$λ$-models: Effective Decision-Aware Reinforcement Learning with Latent Models","date":"2023-06-30","arxiv_id":"2306.17366","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigation-of-micro-robot-swarms-for-targeted","title":"Navigation of micro-robot swarms for targeted delivery using reinforcement learning","date":"2023-06-30","arxiv_id":"2306.17598","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-actor-free-policy-via-convex","title":"Risk-sensitive Actor-free Policy via Convex Optimization","date":"2023-06-30","arxiv_id":"2307.00141","repositories_listed":0,"syntology":null},{"url":null,"slug":"arraybot-reinforcement-learning-for","title":"ArrayBot: Reinforcement Learning for Generalizable Distributed Manipulation through Touch","date":"2023-06-29","arxiv_id":"2306.16857","repositories_listed":0,"syntology":null},{"url":null,"slug":"laxity-aware-scalable-reinforcement-learning","title":"Laxity-Aware Scalable Reinforcement Learning for HVAC Control","date":"2023-06-29","arxiv_id":"2306.16619","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-constraint-for-safety-critical","title":"Probabilistic Constraint for Safety-Critical Reinforcement Learning","date":"2023-06-29","arxiv_id":"2306.17279","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-aware-task-composition-for-discrete","title":"Safety-Aware Task Composition for Discrete and Continuous Reinforcement Learning","date":"2023-06-29","arxiv_id":"2306.17033","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-and-trajectory-planning-for-urban","title":"Action and Trajectory Planning for Urban Autonomous Driving with Hierarchical Reinforcement Learning","date":"2023-06-28","arxiv_id":"2306.15968","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluation-of-reinforcement-learning","title":"Evaluation of Reinforcement Learning Techniques for Trading on a Diverse Portfolio","date":"2023-06-28","arxiv_id":"2309.03202","repositories_listed":0,"syntology":null},{"url":null,"slug":"mastering-nordschleife-a-comprehensive-race","title":"Mastering Nordschleife -- A comprehensive race simulation for AI strategy decision-making in motorsports","date":"2023-06-28","arxiv_id":"2306.16088","repositories_listed":0,"syntology":null},{"url":null,"slug":"recent-advances-in-optimal-transport-for","title":"Recent Advances in Optimal Transport for Machine Learning","date":"2023-06-28","arxiv_id":"2306.16156","repositories_listed":0,"syntology":null},{"url":null,"slug":"sharper-model-free-reinforcement-learning-for","title":"Sharper Model-free Reinforcement Learning for Average-reward Markov Decision Processes","date":"2023-06-28","arxiv_id":"2306.16394","repositories_listed":0,"syntology":null},{"url":null,"slug":"structure-in-reinforcement-learning-a-survey","title":"Structure in Deep Reinforcement Learning: A Survey and Open Problems","date":"2023-06-28","arxiv_id":"2306.16021","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-credit-limit-adjustments-under","title":"Optimizing Credit Limit Adjustments Under Adversarial Goals Using Reinforcement Learning","date":"2023-06-27","arxiv_id":"2306.15585","repositories_listed":0,"syntology":null},{"url":null,"slug":"prioritized-trajectory-replay-a-replay-memory","title":"Prioritized Trajectory Replay: A Replay Memory for Data-driven Reinforcement Learning","date":"2023-06-27","arxiv_id":"2306.15503","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-aware-importance-weighting-for-off","title":"Value-aware Importance Weighting for Off-policy Reinforcement Learning","date":"2023-06-27","arxiv_id":"2306.15625","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-dynamic-programming","title":"Beyond dynamic programming","date":"2023-06-26","arxiv_id":"2306.15029","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-multi-robot-formation-control","title":"Decentralized Multi-Robot Formation Control Using Reinforcement Learning","date":"2023-06-26","arxiv_id":"2306.14489","repositories_listed":0,"syntology":null},{"url":null,"slug":"estimating-player-completion-rate-in-mobile","title":"Estimating player completion rate in mobile puzzle games using reinforcement learning","date":"2023-06-26","arxiv_id":"2306.14626","repositories_listed":0,"syntology":null},{"url":null,"slug":"supervised-pretraining-can-learn-in-context","title":"Supervised Pretraining Can Learn In-Context Reinforcement Learning","date":"2023-06-26","arxiv_id":"2306.14892","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-rlhf-more-difficult-than-standard-rl","title":"Is RLHF More Difficult than Standard RL?","date":"2023-06-25","arxiv_id":"2306.14111","repositories_listed":0,"syntology":null},{"url":null,"slug":"safety-critical-scenario-generation-via","title":"Safety-Critical Scenario Generation Via Reinforcement Learning Based Editing","date":"2023-06-25","arxiv_id":"2306.14131","repositories_listed":0,"syntology":null},{"url":null,"slug":"fighting-uncertainty-with-gradients-offline","title":"Fighting Uncertainty with Gradients: Offline Reinforcement Learning via Diffusion Score Matching","date":"2023-06-24","arxiv_id":"2306.14079","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for-13","title":"Multi-agent Deep Reinforcement Learning for Distributed Load Restoration","date":"2023-06-24","arxiv_id":"2306.14018","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-policy-evaluation-for-reinforcement","title":"Offline Policy Evaluation for Reinforcement Learning with Adaptively Collected Data","date":"2023-06-24","arxiv_id":"2306.14063","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-dead-ends","title":"Safe Reinforcement Learning with Dead-Ends Avoidance and Recovery","date":"2023-06-24","arxiv_id":"2306.13944","repositories_listed":0,"syntology":null},{"url":null,"slug":"waypoint-transformer-reinforcement-learning","title":"Waypoint Transformer: Reinforcement Learning via Supervised Learning with Intermediate Targets","date":"2023-06-24","arxiv_id":"2306.14069","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-coverage-for-pac-reinforcement","title":"Active Coverage for PAC Reinforcement Learning","date":"2023-06-23","arxiv_id":"2306.13601","repositories_listed":0,"syntology":null},{"url":null,"slug":"clue-calibrated-latent-guidance-for-offline","title":"CLUE: Calibrated Latent Guidance for Offline Reinforcement Learning","date":"2023-06-23","arxiv_id":"2306.13412","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-skill-graph-osg-a-framework-for","title":"Offline Skill Graph (OSG): A Framework for Learning and Planning using Offline Reinforcement Learning Skills","date":"2023-06-23","arxiv_id":"2306.13630","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-temporal-logic-2","title":"Reinforcement Learning with Temporal-Logic-Based Causal Diagrams","date":"2023-06-23","arxiv_id":"2306.13732","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-multi-agent-reinforcement-5","title":"Decentralized Multi-Agent Reinforcement Learning with Global State Prediction","date":"2023-06-22","arxiv_id":"2306.12926","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-visual-observation-via-offline-1","title":"Learning from Visual Observation via Offline Pretrained State-to-Go Transformer","date":"2023-06-22","arxiv_id":"2306.12860","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-driving-with-deep-reinforcement","title":"Autonomous Driving with Deep Reinforcement Learning in CARLA Simulation","date":"2023-06-20","arxiv_id":"2306.11217","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-strategy-guided-reinforcement","title":"Evolutionary Strategy Guided Reinforcement Learning via MultiBuffer Communication","date":"2023-06-20","arxiv_id":"2306.11535","repositories_listed":0,"syntology":null},{"url":null,"slug":"int-hrl-towards-intention-based-hierarchical","title":"Int-HRL: Towards Intention-based Hierarchical Reinforcement Learning","date":"2023-06-20","arxiv_id":"2306.11483","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-profitable-nft-image-diffusions-via","title":"Learning Profitable NFT Image Diffusions via Multiple Visual-Policy Guided Reinforcement Learning","date":"2023-06-20","arxiv_id":"2306.11731","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-virtual-fixtures","title":"Reinforcement Learning-based Virtual Fixtures for Teleoperation of Hydraulic Construction Machine","date":"2023-06-20","arxiv_id":"2306.11897","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-shaping-via-diffusion-process-in","title":"Reward Shaping via Diffusion Process in Reinforcement Learning","date":"2023-06-20","arxiv_id":"2306.11885","repositories_listed":0,"syntology":null},{"url":null,"slug":"benchmarking-robustness-of-deep-reinforcement","title":"Benchmarking Robustness of Deep Reinforcement Learning approaches to Online Portfolio Management","date":"2023-06-19","arxiv_id":"2306.10950","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-esg-financial","title":"Deep Reinforcement Learning for ESG financial portfolio management","date":"2023-06-19","arxiv_id":"2307.09631","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-tick-level-data-and-periodical","title":"Integrating Tick-level Data and Periodical Signal for High-frequency Market Making","date":"2023-06-19","arxiv_id":"2306.17179","repositories_listed":0,"syntology":null},{"url":null,"slug":"larg-language-based-automatic-reward-and-goal","title":"LARG, Language-based Automatic Reward and Goal Generation","date":"2023-06-19","arxiv_id":"2306.10985","repositories_listed":0,"syntology":null},{"url":null,"slug":"least-square-value-iteration-is-robust-under","title":"On the Model-Misspecification in Reinforcement Learning","date":"2023-06-19","arxiv_id":"2306.10694","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-execution-using-reinforcement","title":"Optimal Execution Using Reinforcement Learning","date":"2023-06-19","arxiv_id":"2306.17178","repositories_listed":0,"syntology":null},{"url":null,"slug":"vanishing-bias-heuristic-guided-reinforcement","title":"Vanishing Bias Heuristic-guided Reinforcement Learning Algorithm","date":"2023-06-17","arxiv_id":"2306.10216","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-sequential-optimal-experimental","title":"Variational Sequential Optimal Experimental Design using Reinforcement Learning","date":"2023-06-17","arxiv_id":"2306.10430","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-deduction-path-learning-via","title":"Automatic Deduction Path Learning via Reinforcement Learning with Environmental Correction","date":"2023-06-16","arxiv_id":"2306.10083","repositories_listed":0,"syntology":null},{"url":null,"slug":"bootstrapped-representations-in-reinforcement","title":"Bootstrapped Representations in Reinforcement Learning","date":"2023-06-16","arxiv_id":"2306.10171","repositories_listed":0,"syntology":null},{"url":null,"slug":"fairness-in-preference-based-reinforcement","title":"Fairness in Preference-based Reinforcement Learning","date":"2023-06-16","arxiv_id":"2306.09995","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-generative-flow-networks-with","title":"Meta Generative Flow Networks with Personalization for Task-Specific Adaptation","date":"2023-06-16","arxiv_id":"2306.09742","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-evolution-theory-of-learning-from-natural","title":"The Evolution theory of Learning: From Natural Selection to Reinforcement Learning","date":"2023-06-16","arxiv_id":"2306.09961","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-false-dawn-reevaluating-google-s","title":"The False Dawn: Reevaluating Google's Reinforcement Learning for Chip Macro Placement","date":"2023-06-16","arxiv_id":"2306.09633","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-based-open-ran-slice-management","title":"Attention-based Open RAN Slice Management using Deep Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09490","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-generative-models-for-decision-making","title":"Deep Generative Models for Decision-Making and Control","date":"2023-06-15","arxiv_id":"2306.08810","repositories_listed":0,"syntology":null},{"url":null,"slug":"diarel-reinforcement-learning-with","title":"DiAReL: Reinforcement Learning with Disturbance Awareness for Robust Sim2Real Policy Transfer in Robot Control","date":"2023-06-15","arxiv_id":"2306.09010","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-management-for-a-dm-i-plug-in-hybrid","title":"Plug-in Hybrid Electric Vehicle Energy Management with Clutch Engagement Control via Continuous-Discrete Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.08823","repositories_listed":0,"syntology":null},{"url":null,"slug":"granger-causal-hierarchical-skill-discovery","title":"Granger Causal Interaction Skill Chains","date":"2023-06-15","arxiv_id":"2306.09509","repositories_listed":0,"syntology":null},{"url":null,"slug":"inroads-into-autonomous-network-defence-using","title":"Inroads into Autonomous Network Defence using Explained Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.09318","repositories_listed":0,"syntology":null},{"url":null,"slug":"langevin-thompson-sampling-with-logarithmic","title":"Langevin Thompson Sampling with Logarithmic Communication: Bandits and Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.08803","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-multi-agent-reinforcement-learning","title":"Offline Multi-Agent Reinforcement Learning with Coupled Value Factorization","date":"2023-06-15","arxiv_id":"2306.08900","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-maneuver-planning-with-deep","title":"Predictive Maneuver Planning with Deep Reinforcement Learning (PMP-DRL) for comfortable and safe autonomous driving","date":"2023-06-15","arxiv_id":"2306.09055","repositories_listed":0,"syntology":null}],"record_sha256":"1a4ae375fe40be38df69261753cb7e95f9c3850ae07f3cba640dd01eea50eda9","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}