{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/74","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":74,"pages_in_order":135,"rows_per_page":100,"rows":[7301,7400],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/73","next":"/task/reinforcement-learning-2/papers/75","papers":[{"url":null,"slug":"identifying-coordination-in-a-cognitive-radar","title":"Identifying Coordination in a Cognitive Radar Network -- A Multi-Objective Inverse Reinforcement Learning Approach","date":"2022-11-13","arxiv_id":"2211.06967","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for-11","title":"Multi-Agent Deep Reinforcement Learning for Efficient Passenger Delivery in Urban Air Mobility","date":"2022-11-13","arxiv_id":"2211.06890","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-vector","title":"Deep Reinforcement Learning with Vector Quantized Encoding","date":"2022-11-12","arxiv_id":"2211.06733","repositories_listed":0,"syntology":null},{"url":null,"slug":"rewards-encoding-environment-dynamics","title":"Rewards Encoding Environment Dynamics Improves Preference-based Reinforcement Learning","date":"2022-11-12","arxiv_id":"2211.06527","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-commercial-cooling-systems-using","title":"Controlling Commercial Cooling Systems Using Reinforcement Learning","date":"2022-11-11","arxiv_id":"2211.07357","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-microgrid","title":"Deep Reinforcement Learning Microgrid Optimization Strategy Considering Priority Flexible Demand Side","date":"2022-11-11","arxiv_id":"2211.05946","repositories_listed":0,"syntology":null},{"url":null,"slug":"emergency-action-termination-for-immediate","title":"Emergency action termination for immediate reaction in hierarchical reinforcement learning","date":"2022-11-11","arxiv_id":"2211.06351","repositories_listed":0,"syntology":null},{"url":null,"slug":"gradient-imitation-reinforcement-learning-for-1","title":"Gradient Imitation Reinforcement Learning for General Low-Resource Information Extraction","date":"2022-11-11","arxiv_id":"2211.06014","repositories_listed":0,"syntology":null},{"url":null,"slug":"job-scheduling-in-datacenters-using","title":"Job Scheduling in Datacenters using Constraint Controlled RL","date":"2022-11-10","arxiv_id":"2211.05338","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-n-1-secure-hv-grid-flexibility","title":"Robust N-1 secure HV Grid Flexibility Estimation for TSO-DSO coordinated Congestion Management with Deep Reinforcement Learning","date":"2022-11-10","arxiv_id":"2211.05855","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-is-realizability-sufficient-for-off","title":"When is Realizability Sufficient for Off-Policy Reinforcement Learning?","date":"2022-11-10","arxiv_id":"2211.05311","repositories_listed":0,"syntology":null},{"url":null,"slug":"foundation-models-for-semantic-novelty-in","title":"Foundation Models for Semantic Novelty in Reinforcement Learning","date":"2022-11-09","arxiv_id":"2211.04878","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-deep-reinforcement-learning-for","title":"Interpretable Deep Reinforcement Learning for Green Security Games with Real-Time Information","date":"2022-11-09","arxiv_id":"2211.04987","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-offline-data-in-online","title":"Leveraging Offline Data in Online Reinforcement Learning","date":"2022-11-09","arxiv_id":"2211.04974","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-agent-to-multi-agent-in-deep","title":"Solving Collaborative Dec-POMDPs with Deep Reinforcement Learning Heuristics","date":"2022-11-09","arxiv_id":"2211.15411","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-compressed-ratio-estimation-using","title":"Efficient Compressed Ratio Estimation Using Online Sequential Learning for Edge Computing","date":"2022-11-08","arxiv_id":"2211.04284","repositories_listed":0,"syntology":null},{"url":null,"slug":"pretraining-in-deep-reinforcement-learning-a","title":"Pretraining in Deep Reinforcement Learning: A Survey","date":"2022-11-08","arxiv_id":"2211.03959","repositories_listed":0,"syntology":null},{"url":null,"slug":"progress-and-summary-of-reinforcement","title":"Progress and summary of reinforcement learning on energy management of MPS-EV","date":"2022-11-08","arxiv_id":"2211.04001","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-stepwise-fairness","title":"Reinforcement Learning with Stepwise Fairness Constraints","date":"2022-11-08","arxiv_id":"2211.03994","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-quantum-reinforcement-learning","title":"A Survey on Quantum Reinforcement Learning","date":"2022-11-07","arxiv_id":"2211.03464","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-predictive-clustering","title":"Reward-Predictive Clustering","date":"2022-11-07","arxiv_id":"2211.03281","repositories_listed":0,"syntology":null},{"url":null,"slug":"exposing-surveillance-detection-routes-via","title":"Exposing Surveillance Detection Routes via Reinforcement Learning, Attack Graphs, and Cyber Terrain","date":"2022-11-06","arxiv_id":"2211.03027","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatio-temporal-incentives-optimization-for","title":"Spatio-temporal Incentives Optimization for Ride-hailing Services with Offline Deep Reinforcement Learning","date":"2022-11-06","arxiv_id":"2211.03240","repositories_listed":0,"syntology":null},{"url":null,"slug":"wall-street-tree-search-risk-aware-planning","title":"Wall Street Tree Search: Risk-Aware Planning for Offline Reinforcement Learning","date":"2022-11-06","arxiv_id":"2211.04583","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-federated-reinforcement","title":"Decentralized Federated Reinforcement Learning for User-Centric Dynamic TFDD Control","date":"2022-11-04","arxiv_id":"2211.02296","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-reinforcement-learning-in","title":"A Survey on Reinforcement Learning in Aviation Applications","date":"2022-11-03","arxiv_id":"2211.02147","repositories_listed":0,"syntology":null},{"url":"/paper/lilgym-natural-language-visual-reasoning-with","slug":"lilgym-natural-language-visual-reasoning-with","title":"lilGym: Natural Language Visual Reasoning with Reinforcement Learning","date":"2022-11-03","arxiv_id":"2211.01994","repositories_listed":0,"syntology":null},{"url":null,"slug":"oracle-inequalities-for-model-selection-in","title":"Oracle Inequalities for Model Selection in Offline Reinforcement Learning","date":"2022-11-03","arxiv_id":"2211.02016","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-non-markovian","title":"Reinforcement Learning in Non-Markovian Environments","date":"2022-11-03","arxiv_id":"2211.01595","repositories_listed":0,"syntology":null},{"url":null,"slug":"theta-resonance-a-single-step-reinforcement","title":"Theta-Resonance: A Single-Step Reinforcement Learning Method for Design Space Exploration","date":"2022-11-03","arxiv_id":"2211.02052","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-irs-phase","title":"Deep Reinforcement Learning for IRS Phase Shift Design in Spatiotemporally Correlated Environments","date":"2022-11-02","arxiv_id":"2211.09726","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-power-control","title":"Deep Reinforcement Learning for Power Control in Next-Generation WiFi Network Systems","date":"2022-11-02","arxiv_id":"2211.01107","repositories_listed":0,"syntology":null},{"url":"/paper/dual-generator-offline-reinforcement-learning","slug":"dual-generator-offline-reinforcement-learning","title":"Dual Generator Offline Reinforcement Learning","date":"2022-11-02","arxiv_id":"2211.01471","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dual-generator-offline-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2211.01471","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.01471"}},"official":null}},{"url":null,"slug":"model-based-reinforcement-learning-with-a-1","title":"Model-based Reinforcement Learning with a Hamiltonian Canonical ODE Network","date":"2022-11-02","arxiv_id":"2211.00942","repositories_listed":0,"syntology":null},{"url":null,"slug":"wind-power-forecasting-considering-data","title":"Wind Power Forecasting Considering Data Privacy Protection: A Federated Deep Reinforcement Learning Approach","date":"2022-11-02","arxiv_id":"2211.02674","repositories_listed":0,"syntology":null},{"url":null,"slug":"discrete-factorial-representations-as-an","title":"Discrete Factorial Representations as an Abstraction for Goal Conditioned Reinforcement Learning","date":"2022-11-01","arxiv_id":"2211.00247","repositories_listed":0,"syntology":null},{"url":null,"slug":"event-tables-for-efficient-experience-replay","title":"Event Tables for Efficient Experience Replay","date":"2022-11-01","arxiv_id":"2211.00576","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-applied-to-trading","title":"Reinforcement Learning Applied to Trading Systems: A Survey","date":"2022-11-01","arxiv_id":"2212.06064","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-education-a-multi","title":"Reinforcement Learning in Education: A Multi-Armed Bandit Approach","date":"2022-11-01","arxiv_id":"2211.00779","repositories_listed":0,"syntology":null},{"url":null,"slug":"danzero-mastering-guandan-game-with","title":"DanZero: Mastering GuanDan Game with Reinforcement Learning","date":"2022-10-31","arxiv_id":"2210.17087","repositories_listed":0,"syntology":null},{"url":null,"slug":"teacher-student-curriculum-learning-for","title":"Teacher-student curriculum learning for reinforcement learning","date":"2022-10-31","arxiv_id":"2210.17368","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-rate-distortion-theory-in-capacity-limited","title":"On Rate-Distortion Theory in Capacity-Limited Cognition & Reinforcement Learning","date":"2022-10-30","arxiv_id":"2210.16877","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-to-the-information-horizon-of-bamdps","title":"Planning to the Information Horizon of BAMDPs via Epistemic State Abstraction","date":"2022-10-30","arxiv_id":"2210.16872","repositories_listed":0,"syntology":null},{"url":null,"slug":"learninggroup-a-real-time-sparse-training-on","title":"LearningGroup: A Real-Time Sparse Training on FPGA via Learnable Weight Grouping for Multi-Agent Reinforcement Learning","date":"2022-10-29","arxiv_id":"2210.16624","repositories_listed":0,"syntology":null},{"url":null,"slug":"nonuniqueness-and-convergence-to-equivalent","title":"Nonuniqueness and Convergence to Equivalent Solutions in Observer-based Inverse Reinforcement Learning","date":"2022-10-28","arxiv_id":"2210.16299","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-defect","title":"Reinforcement Learning-based Defect Mitigation for Quality Assurance of Additive Manufacturing","date":"2022-10-28","arxiv_id":"2210.17272","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-contrastive-samples-for-identifying-and","title":"Using Contrastive Samples for Identifying and Leveraging Possible Causal Relationships in Reinforcement Learning","date":"2022-10-28","arxiv_id":"2210.17296","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-indoor-localization-via-reinforcement","title":"Hybrid Indoor Localization via Reinforcement Learning-based Information Fusion","date":"2022-10-27","arxiv_id":"2210.15132","repositories_listed":0,"syntology":null},{"url":null,"slug":"many-objective-reinforcement-learning-for","title":"Many-Objective Reinforcement Learning for Online Testing of DNN-Enabled Systems","date":"2022-10-27","arxiv_id":"2210.15432","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-using-model","title":"Meta-Reinforcement Learning Using Model Parameters","date":"2022-10-27","arxiv_id":"2210.15515","repositories_listed":0,"syntology":null},{"url":null,"slug":"sam-rl-sensing-aware-model-based","title":"SAM-RL: Sensing-Aware Model-Based Reinforcement Learning via Differentiable Physics-Based Simulation and Rendering","date":"2022-10-27","arxiv_id":"2210.15185","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-customizable-reinforcement-learning","title":"Towards customizable reinforcement learning agents: Enabling preference specification through online vocabulary expansion","date":"2022-10-27","arxiv_id":"2210.15096","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bibliometric-analysis-and-review-on","title":"A Bibliometric Analysis and Review on Reinforcement Learning for Transportation Applications","date":"2022-10-26","arxiv_id":"2210.14524","repositories_listed":0,"syntology":null},{"url":null,"slug":"d-shape-demonstration-shaped-reinforcement","title":"D-Shape: Demonstration-Shaped Reinforcement Learning via Goal Conditioning","date":"2022-10-26","arxiv_id":"2210.14428","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-guided-exploration-in-deep","title":"Knowledge-Guided Exploration in Deep Reinforcement Learning","date":"2022-10-26","arxiv_id":"2210.15670","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-deep-recurrent-reinforcement-learning","title":"Quantum deep recurrent reinforcement learning","date":"2022-10-26","arxiv_id":"2210.14876","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-based-meta-reinforcement-learning","title":"Uncertainty-based Meta-Reinforcement Learning for Robust Radar Tracking","date":"2022-10-26","arxiv_id":"2210.14532","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-distributional-and-risk-sensitive","title":"Bridging Distributional and Risk-sensitive Reinforcement Learning with Provable Regret Bounds","date":"2022-10-25","arxiv_id":"2210.14051","repositories_listed":0,"syntology":null},{"url":null,"slug":"entity-divider-with-language-grounding-in","title":"Entity Divider with Language Grounding in Multi-Agent Reinforcement Learning","date":"2022-10-25","arxiv_id":"2210.13942","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-shot-offline-and-production-scalable-pid","title":"One-shot, Offline and Production-Scalable PID Optimisation with Deep Reinforcement Learning","date":"2022-10-25","arxiv_id":"2210.13906","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-explanation-for-reinforcement-learning","title":"Causal Explanation for Reinforcement Learning: Quantifying State and Temporal Importance","date":"2022-10-24","arxiv_id":"2210.13507","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-reinforcement-learning-based-cnn","title":"Graph Reinforcement Learning-based CNN Inference Offloading in Dynamic Edge Computing","date":"2022-10-24","arxiv_id":"2210.13464","repositories_listed":0,"syntology":null},{"url":null,"slug":"hardness-in-markov-decision-processes-theory","title":"Hardness in Markov Decision Processes: Theory and Practice","date":"2022-10-24","arxiv_id":"2210.13075","repositories_listed":0,"syntology":null},{"url":null,"slug":"opportunistic-episodic-reinforcement-learning","title":"Opportunistic Episodic Reinforcement Learning","date":"2022-10-24","arxiv_id":"2210.13504","repositories_listed":0,"syntology":null},{"url":null,"slug":"oss-mentor-a-framework-for-improving","title":"OSS Mentor A framework for improving developers contributions via deep reinforcement learning","date":"2022-10-24","arxiv_id":"2210.13990","repositories_listed":0,"syntology":null},{"url":null,"slug":"reachability-aware-laplacian-representation","title":"Reachability-Aware Laplacian Representation in Reinforcement Learning","date":"2022-10-24","arxiv_id":"2210.13153","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-and-bandits-for-speech","title":"Reinforcement Learning and Bandits for Speech and Language Processing: Tutorial, Review and Outlook","date":"2022-10-24","arxiv_id":"2210.13623","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-cooperative-reinforcement-learning","title":"A Cooperative Reinforcement Learning Environment for Detecting and Penalizing Betrayal","date":"2022-10-23","arxiv_id":"2210.12841","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-predictive-coding-a-unified-neural","title":"Active Predictive Coding: A Unified Neural Framework for Learning Hierarchical World Models for Perception and Planning","date":"2022-10-23","arxiv_id":"2210.13461","repositories_listed":0,"syntology":null},{"url":null,"slug":"climate-change-policy-exploration-using","title":"Climate Change Policy Exploration using Reinforcement Learning","date":"2022-10-23","arxiv_id":"2211.17013","repositories_listed":0,"syntology":null},{"url":null,"slug":"metaems-a-meta-reinforcement-learning-based","title":"MetaEMS: A Meta Reinforcement Learning-based Control Framework for Building Energy Management System","date":"2022-10-23","arxiv_id":"2210.12590","repositories_listed":0,"syntology":null},{"url":null,"slug":"faster-and-more-diverse-de-novo-molecular","title":"Faster and more diverse de novo molecular optimization with double-loop reinforcement learning using augmented SMILES","date":"2022-10-22","arxiv_id":"2210.12458","repositories_listed":0,"syntology":null},{"url":null,"slug":"probing-transfer-in-deep-reinforcement","title":"Probing Transfer in Deep Reinforcement Learning without Task Engineering","date":"2022-10-22","arxiv_id":"2210.12448","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-reinforcement-learning-with-group","title":"Continual Vision-based Reinforcement Learning with Group Symmetries","date":"2022-10-21","arxiv_id":"2210.12301","repositories_listed":0,"syntology":null},{"url":null,"slug":"counterfactual-explanations-for-reinforcement","title":"Redefining Counterfactual Explanations for Reinforcement Learning: Overview, Challenges and Opportunities","date":"2022-10-21","arxiv_id":"2210.11846","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-inverse","title":"Deep Reinforcement Learning for Inverse Inorganic Materials Design","date":"2022-10-21","arxiv_id":"2210.11931","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-stabilization","title":"Deep Reinforcement Learning for Stabilization of Large-scale Probabilistic Boolean Networks","date":"2022-10-21","arxiv_id":"2210.12229","repositories_listed":0,"syntology":null},{"url":null,"slug":"group-distributionally-robust-reinforcement","title":"Group Distributionally Robust Reinforcement Learning with Hierarchical Latent Variables","date":"2022-10-21","arxiv_id":"2210.12262","repositories_listed":0,"syntology":null},{"url":null,"slug":"implicit-offline-reinforcement-learning-via","title":"Implicit Offline Reinforcement Learning via Supervised Learning","date":"2022-10-21","arxiv_id":"2210.12272","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-policy-summaries-with-reward","title":"Integrating Policy Summaries with Reward Decomposition for Explaining Reinforcement Learning Agents","date":"2022-10-21","arxiv_id":"2210.11825","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-connection-between-bregman-divergence","title":"On the connection between Bregman divergence and value in regularized Markov decision processes","date":"2022-10-21","arxiv_id":"2210.12160","repositories_listed":0,"syntology":null},{"url":null,"slug":"planning-with-uncertainty-deep-exploration-in","title":"Epistemic Monte Carlo Tree Search","date":"2022-10-21","arxiv_id":"2210.13455","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-quantum-enabled-6g-slicing","title":"Towards Quantum-Enabled 6G Slicing","date":"2022-10-21","arxiv_id":"2212.11755","repositories_listed":0,"syntology":null},{"url":null,"slug":"fine-grained-session-recommendations-in-e","title":"Fine-Grained Session Recommendations in E-commerce using Deep Reinforcement Learning","date":"2022-10-20","arxiv_id":"2210.15451","repositories_listed":0,"syntology":null},{"url":null,"slug":"horizon-free-reinforcement-learning-for","title":"Horizon-Free and Variance-Dependent Reinforcement Learning for Latent Markov Decision Processes","date":"2022-10-20","arxiv_id":"2210.11604","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-imitation-via-mirror-descent-inverse-1","title":"Robust Imitation via Mirror Descent Inverse Reinforcement Learning","date":"2022-10-20","arxiv_id":"2210.11201","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-in-multi","title":"A Reinforcement Learning Approach in Multi-Phase Second-Price Auction Design","date":"2022-10-19","arxiv_id":"2210.10278","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-7","title":"Hierarchical Reinforcement Learning for Furniture Layout in Virtual Indoor Scenes","date":"2022-10-19","arxiv_id":"2210.10431","repositories_listed":0,"syntology":null},{"url":null,"slug":"oracles-followers-stackelberg-equilibria-in","title":"Oracles & Followers: Stackelberg Equilibria in Deep Multi-Agent Reinforcement Learning","date":"2022-10-19","arxiv_id":"2210.11942","repositories_listed":0,"syntology":null},{"url":null,"slug":"palm-up-playing-in-the-latent-manifold-for","title":"Palm up: Playing in the Latent Manifold for Unsupervised Pretraining","date":"2022-10-19","arxiv_id":"2210.10913","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-safe-reinforcement-learning-via","title":"Provably Safe Reinforcement Learning via Action Projection using Reachability Analysis and Polynomial Zonotopes","date":"2022-10-19","arxiv_id":"2210.10691","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-table-wiping-via-reinforcement","title":"Robotic Table Wiping via Reinforcement Learning and Whole-body Trajectory Optimization","date":"2022-10-19","arxiv_id":"2210.10865","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-laws-for-reward-model","title":"Scaling Laws for Reward Model Overoptimization","date":"2022-10-19","arxiv_id":"2210.10760","repositories_listed":0,"syntology":null},{"url":null,"slug":"rpm-generalizable-behaviors-for-multi-agent","title":"RPM: Generalizable Behaviors for Multi-Agent Reinforcement Learning","date":"2022-10-18","arxiv_id":"2210.09646","repositories_listed":0,"syntology":null},{"url":null,"slug":"unpacking-reward-shaping-understanding-the","title":"Unpacking Reward Shaping: Understanding the Benefits of Reward Engineering on Sample Complexity","date":"2022-10-18","arxiv_id":"2210.09579","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-offline-reinforcement-learning-via","title":"Boosting Offline Reinforcement Learning via Data Rebalancing","date":"2022-10-17","arxiv_id":"2210.09241","repositories_listed":0,"syntology":null},{"url":null,"slug":"ptde-personalized-training-with-distillated","title":"PTDE: Personalized Training with Distilled Execution for Multi-Agent Reinforcement Learning","date":"2022-10-17","arxiv_id":"2210.08872","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-only-live-once-single-life-reinforcement","title":"You Only Live Once: Single-Life Reinforcement Learning","date":"2022-10-17","arxiv_id":"2210.08863","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-pipeline-for-offline","title":"Data-Efficient Pipeline for Offline Reinforcement Learning with Limited Data","date":"2022-10-16","arxiv_id":"2210.08642","repositories_listed":0,"syntology":null},{"url":null,"slug":"entropy-regularized-reinforcement-learning","title":"Entropy Regularized Reinforcement Learning with Cascading Networks","date":"2022-10-16","arxiv_id":"2210.08503","repositories_listed":0,"syntology":null}],"record_sha256":"ebc7951a3bd5330f8db5817905161dd231391d685a97b51da58db131bf0ec978","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}