{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/70","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":70,"pages_in_order":135,"rows_per_page":100,"rows":[6901,7000],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/69","next":"/task/reinforcement-learning-2/papers/71","papers":[{"url":null,"slug":"sim-and-real-reinforcement-learning-for","title":"Sim-and-Real Reinforcement Learning for Manipulation: A Consensus-based Approach","date":"2023-02-26","arxiv_id":"2302.13423","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-human-centered-safe-robot-reinforcement","title":"A Human-Centered Safe Robot Reinforcement Learning Framework with Interactive Behaviors","date":"2023-02-25","arxiv_id":"2302.13137","repositories_listed":0,"syntology":null},{"url":null,"slug":"exponential-hardness-of-reinforcement","title":"Exponential Hardness of Reinforcement Learning with Linear Function Approximation","date":"2023-02-25","arxiv_id":"2302.12940","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-bellman-s-principle-of-optimality-and","title":"On Bellman's principle of optimality and Reinforcement learning for safety-constrained Markov decision process","date":"2023-02-25","arxiv_id":"2302.13152","repositories_listed":0,"syntology":null},{"url":null,"slug":"ac2c-adaptively-controlled-two-hop","title":"AC2C: Adaptively Controlled Two-Hop Communication for Multi-Agent Reinforcement Learning","date":"2023-02-24","arxiv_id":"2302.12515","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-regularized-competitive-equilibria-of","title":"Finding Regularized Competitive Equilibria of Heterogeneous Agent Macroeconomic Models with Reinforcement Learning","date":"2023-02-24","arxiv_id":"2303.04833","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-jumpy-models-for-planning-and-fast","title":"Leveraging Jumpy Models for Planning and Fast Learning in Robotic Domains","date":"2023-02-24","arxiv_id":"2302.12617","repositories_listed":0,"syntology":null},{"url":null,"slug":"logarithmic-switching-cost-in-reinforcement","title":"Logarithmic Switching Cost in Reinforcement Learning beyond Linear MDPs","date":"2023-02-24","arxiv_id":"2302.12456","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-with-5","title":"Multi-Agent Reinforcement Learning with Common Policy for Antenna Tilt Optimization","date":"2023-02-24","arxiv_id":"2302.12899","repositories_listed":0,"syntology":null},{"url":null,"slug":"concept-learning-for-interpretable-multi","title":"Concept Learning for Interpretable Multi-Agent Reinforcement Learning","date":"2023-02-23","arxiv_id":"2302.12232","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-reinforcement-learning-via","title":"Provably Efficient Reinforcement Learning via Surprise Bound","date":"2023-02-22","arxiv_id":"2302.11634","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-framework-for-online","title":"A Reinforcement Learning Framework for Online Speaker Diarization","date":"2023-02-21","arxiv_id":"2302.10924","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-model-for-offline-reinforcement","title":"Adversarial Model for Offline Reinforcement Learning","date":"2023-02-21","arxiv_id":"2302.11048","repositories_listed":0,"syntology":null},{"url":null,"slug":"badgpt-exploring-security-vulnerabilities-of","title":"BadGPT: Exploring Security Vulnerabilities of ChatGPT via Backdoor Attacks to InstructGPT","date":"2023-02-21","arxiv_id":"2304.12298","repositories_listed":0,"syntology":null},{"url":null,"slug":"conditioning-hierarchical-reinforcement","title":"Handling Long and Richly Constrained Tasks through Constrained Hierarchical Reinforcement Learning","date":"2023-02-21","arxiv_id":"2302.10639","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-reinforcement-learning-for-3","title":"Constrained Reinforcement Learning for Predictive Control in Real-Time Stochastic Dynamic Optimal Power Flow","date":"2023-02-21","arxiv_id":"2302.10382","repositories_listed":0,"syntology":null},{"url":null,"slug":"curiosity-driven-exploration-in-sparse-reward","title":"Curiosity-driven Exploration in Sparse-reward Multi-agent Reinforcement Learning","date":"2023-02-21","arxiv_id":"2302.10825","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-robotic-2","title":"Deep Reinforcement Learning for Robotic Pushing and Picking in Cluttered Environment","date":"2023-02-21","arxiv_id":"2302.10717","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-for-mixture-of","title":"Offline Reinforcement Learning for Mixture-of-Expert Dialogue Management","date":"2023-02-21","arxiv_id":"2302.10850","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-exploration-in-quantum","title":"Provably Efficient Exploration in Quantum Reinforcement Learning with Logarithmic Worst-Case Regret","date":"2023-02-21","arxiv_id":"2302.10796","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-block","title":"Reinforcement Learning for Block Decomposition of CAD Models","date":"2023-02-21","arxiv_id":"2302.11066","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-a-birth-and-death","title":"Reinforcement Learning in a Birth and Death Process: Breaking the Dependence on the State Space","date":"2023-02-21","arxiv_id":"2302.10667","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-auto-landing-control-of-an-agile","title":"Robust Auto-landing Control of an agile Regional Jet Using Fuzzy Q-learning","date":"2023-02-21","arxiv_id":"2302.10997","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-path-planning-employing-mpc-reinforcement","title":"UAV Path Planning Employing MPC- Reinforcement Learning Method Considering Collision Avoidance","date":"2023-02-21","arxiv_id":"2302.10669","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-arbitrating-in-zero-sum-markov","title":"Differentiable Arbitrating in Zero-sum Markov Games","date":"2023-02-20","arxiv_id":"2302.10058","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-function","title":"Reinforcement Learning with Function Approximation: From Linear to Nonlinear","date":"2023-02-20","arxiv_id":"2302.09703","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-deep-reinforcement-learning-by-verifying","title":"Safe Deep Reinforcement Learning by Verifying Task-Level Properties","date":"2023-02-20","arxiv_id":"2302.10030","repositories_listed":0,"syntology":null},{"url":null,"slug":"autodoviz-human-centered-automation-for","title":"AutoDOViz: Human-Centered Automation for Decision Optimization","date":"2023-02-19","arxiv_id":"2302.09688","repositories_listed":0,"syntology":null},{"url":null,"slug":"compositionality-and-bounds-for-optimal-value","title":"Compositionality and Bounds for Optimal Value Functions in Reinforcement Learning","date":"2023-02-19","arxiv_id":"2302.09676","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-video-corpus-moment-retrieval","title":"Interactive Video Corpus Moment Retrieval using Reinforcement Learning","date":"2023-02-19","arxiv_id":"2302.09522","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-and-versatile-bipedal-jumping-control","title":"Robust and Versatile Bipedal Jumping Control through Reinforcement Learning","date":"2023-02-19","arxiv_id":"2302.09450","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-multimodal-reinforcement-learning","title":"Effective Multimodal Reinforcement Learning with Modality Alignment and Importance Enhancement","date":"2023-02-18","arxiv_id":"2302.09318","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-exploration-via-epistemic-risk","title":"Efficient Exploration via Epistemic-Risk-Seeking Policy Optimization","date":"2023-02-18","arxiv_id":"2302.09339","repositories_listed":0,"syntology":null},{"url":null,"slug":"promoting-cooperation-in-multi-agent","title":"Promoting Cooperation in Multi-Agent Reinforcement Learning via Mutual Help","date":"2023-02-18","arxiv_id":"2302.09277","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-the-wild-with","title":"Reinforcement Learning in the Wild with Maximum Likelihood-based Model Transfer","date":"2023-02-18","arxiv_id":"2302.09273","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-state-augmentation-based-approach-to","title":"A State Augmentation based approach to Reinforcement Learning from Human Preferences","date":"2023-02-17","arxiv_id":"2302.08734","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-reward-initialization-for","title":"Data Driven Reward Initialization for Preference based Reinforcement Learning","date":"2023-02-17","arxiv_id":"2302.08733","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-mmwave","title":"Deep Reinforcement Learning for mmWave Initial Beam Alignment","date":"2023-02-17","arxiv_id":"2302.08969","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-unlabeled-data-for-feedback","title":"Exploiting Unlabeled Data for Feedback Efficient Human Preference based Reinforcement Learning","date":"2023-02-17","arxiv_id":"2302.08738","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-forecast-aleatoric-and-epistemic","title":"Learning to Forecast Aleatoric and Epistemic Uncertainties over Long Horizon Trajectories","date":"2023-02-17","arxiv_id":"2302.08669","repositories_listed":0,"syntology":null},{"url":null,"slug":"robot-path-planning-using-deep-reinforcement","title":"Robot path planning using deep reinforcement learning","date":"2023-02-17","arxiv_id":"2302.09120","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-computing-provides-exponential-regret","title":"Quantum Computing Provides Exponential Regret Improvement in Episodic Reinforcement Learning","date":"2023-02-16","arxiv_id":"2302.08617","repositories_listed":0,"syntology":null},{"url":null,"slug":"ceril-continuous-event-based-reinforcement","title":"CERiL: Continuous Event-based Reinforcement Learning","date":"2023-02-15","arxiv_id":"2302.07667","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-offline-reinforcement-learning-for-real","title":"Deep Offline Reinforcement Learning for Real-world Treatment Optimization Applications","date":"2023-02-15","arxiv_id":"2302.07549","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-via-exploratory","title":"Meta-Reinforcement Learning via Exploratory Task Clustering","date":"2023-02-15","arxiv_id":"2302.07958","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-sample-complexity-of-reinforcement","title":"Optimal Sample Complexity of Reinforcement Learning for Mixing Discounted Markov Decision Processes","date":"2023-02-15","arxiv_id":"2302.07477","repositories_listed":0,"syntology":null},{"url":null,"slug":"prioritized-offline-goal-swapping-experience","title":"Prioritized offline Goal-swapping Experience Replay","date":"2023-02-15","arxiv_id":"2302.07741","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-power-grid-day","title":"Reinforcement Learning Based Power Grid Day-Ahead Planning and AI-Assisted Control","date":"2023-02-15","arxiv_id":"2302.07654","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-multi-agent-reinforcement-learning-4","title":"Scalable Multi-Agent Reinforcement Learning with General Utilities","date":"2023-02-15","arxiv_id":"2302.07938","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-multi-user","title":"Deep Reinforcement Learning for Multi-user Massive MIMO with Channel Aging","date":"2023-02-14","arxiv_id":"2302.06853","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-algorithms-applied-to-satellite","title":"Quantum algorithms applied to satellite mission planning for Earth observation","date":"2023-02-14","arxiv_id":"2302.07181","repositories_listed":0,"syntology":null},{"url":null,"slug":"to-risk-or-not-to-risk-learning-with-risk","title":"To Risk or Not to Risk: Learning with Risk Quantification for IoT Task Offloading in UAVs","date":"2023-02-14","arxiv_id":"2302.07399","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-lifetime-extended-energy-management","title":"A Lifetime Extended Energy Management Strategy for Fuel Cell Hybrid Electric Vehicles via Self-Learning Fuzzy Reinforcement Learning","date":"2023-02-13","arxiv_id":"2302.06236","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-safe-reinforcement-learning-with","title":"Provably Safe Reinforcement Learning with Step-wise Violation Constraints","date":"2023-02-13","arxiv_id":"2302.06064","repositories_listed":0,"syntology":null},{"url":null,"slug":"maneuver-decision-making-for-autonomous-air","title":"Maneuver Decision-Making For Autonomous Air Combat Through Curriculum Learning And Reinforcement Learning With Sparse Rewards","date":"2023-02-12","arxiv_id":"2302.05838","repositories_listed":0,"syntology":null},{"url":null,"slug":"remix-regret-minimization-for-monotonic-value","title":"ReMIX: Regret Minimization for Monotonic Value Function Factorization in Multiagent Reinforcement Learning","date":"2023-02-11","arxiv_id":"2302.05593","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-causal-reinforcement-learning","title":"A Survey on Causal Reinforcement Learning","date":"2023-02-10","arxiv_id":"2302.05209","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-entropy-communication-in-multi-agent","title":"Low Entropy Communication in Multi-Agent Reinforcement Learning","date":"2023-02-10","arxiv_id":"2302.05055","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-minimax-optimality-of-model-based","title":"Towards Minimax Optimality of Model-based Robust Reinforcement Learning","date":"2023-02-10","arxiv_id":"2302.05372","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-into-pre-training-object","title":"An Investigation into Pre-Training Object-Centric Representations for Reinforcement Learning","date":"2023-02-09","arxiv_id":"2302.04419","repositories_listed":0,"syntology":null},{"url":null,"slug":"clare-conservative-model-based-reward","title":"CLARE: Conservative Model-Based Reward Learning for Offline Inverse Reinforcement Learning","date":"2023-02-09","arxiv_id":"2302.04782","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-quality-aware-mixed-precision","title":"Data Quality-aware Mixed-precision Quantization via Hybrid Reinforcement Learning","date":"2023-02-09","arxiv_id":"2302.04453","repositories_listed":0,"syntology":null},{"url":null,"slug":"equivariant-muzero","title":"Equivariant MuZero","date":"2023-02-09","arxiv_id":"2302.04798","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-near-optimal-algorithm-for-safe","title":"A Near-Optimal Algorithm for Safe Reinforcement Learning Under Instantaneous Hard Constraints","date":"2023-02-08","arxiv_id":"2302.04375","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-scale-independent-multi-objective","title":"A Scale-Independent Multi-Objective Reinforcement Learning with Convergence Analysis","date":"2023-02-08","arxiv_id":"2302.04179","repositories_listed":0,"syntology":null},{"url":null,"slug":"aisyn-ai-driven-reinforcement-learning-based","title":"AISYN: AI-driven Reinforcement Learning-Based Logic Synthesis Framework","date":"2023-02-08","arxiv_id":"2302.06415","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-planning-in-combinatorial-action","title":"Efficient Planning in Combinatorial Action Spaces with Applications to Cooperative Multi-Agent Reinforcement Learning","date":"2023-02-08","arxiv_id":"2302.04376","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-optimal-adversarial-reinforcement","title":"Near-Optimal Adversarial Reinforcement Learning with Switching Costs","date":"2023-02-08","arxiv_id":"2302.04374","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-aggregation-for-safety-critical","title":"Adaptive Aggregation for Safety-Critical Control","date":"2023-02-07","arxiv_id":"2302.03586","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensemble-value-functions-for-efficient","title":"Ensemble Value Functions for Efficient Exploration in Multi-Agent Reinforcement Learning","date":"2023-02-07","arxiv_id":"2302.03439","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-minimax-optimal-risk-sensitive","title":"Near-Minimax-Optimal Risk-Sensitive Reinforcement Learning with CVaR","date":"2023-02-07","arxiv_id":"2302.03201","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-reinforcement-learning-with-uncertain","title":"Online Reinforcement Learning with Uncertain Episode Lengths","date":"2023-02-07","arxiv_id":"2302.03608","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-audio-recommendations-for-the-long","title":"Optimizing Audio Recommendations for the Long-Term: A Reinforcement Learning Perspective","date":"2023-02-07","arxiv_id":"2302.03561","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-skilled-population-curriculum-for","title":"Towards Skilled Population Curriculum for Multi-Agent Reinforcement Learning","date":"2023-02-07","arxiv_id":"2302.03429","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-for-process-design-with","title":"Transfer learning for process design with reinforcement learning","date":"2023-02-07","arxiv_id":"2302.03375","repositories_listed":0,"syntology":null},{"url":null,"slug":"arena-web-a-web-based-development-and","title":"Arena-Web -- A Web-based Development and Benchmarking Platform for Autonomous Navigation Approaches","date":"2023-02-06","arxiv_id":"2302.02898","repositories_listed":0,"syntology":null},{"url":null,"slug":"ditto-offline-imitation-learning-with-world","title":"DITTO: Offline Imitation Learning with World Models","date":"2023-02-06","arxiv_id":"2302.03086","repositories_listed":0,"syntology":null},{"url":null,"slug":"holistic-deep-reinforcement-learning-based","title":"Holistic Deep-Reinforcement-Learning-based Training of Autonomous Navigation Systems","date":"2023-02-06","arxiv_id":"2302.02921","repositories_listed":0,"syntology":null},{"url":null,"slug":"rltp-reinforcement-learning-to-pace-for","title":"RLTP: Reinforcement Learning to Pace for Delayed Impression Modeling in Preloaded Ads","date":"2023-02-06","arxiv_id":"2302.02592","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-wise-safe-reinforcement-learning-a","title":"State-wise Safe Reinforcement Learning: A Survey","date":"2023-02-06","arxiv_id":"2302.03122","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-online-model-following-projection","title":"An Online Model-Following Projection Mechanism Using Reinforcement Learning","date":"2023-02-05","arxiv_id":"2302.02493","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-problems-and-modern-solutions-for-deep","title":"Open Problems and Modern Solutions for Deep Reinforcement Learning","date":"2023-02-05","arxiv_id":"2302.02298","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-traffic-light-1","title":"Deep Reinforcement Learning for Traffic Light Control in Intelligent Transportation Systems","date":"2023-02-04","arxiv_id":"2302.03669","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalization-of-deep-reinforcement-learning","title":"Generalization of Deep Reinforcement Learning for Jammer-Resilient Frequency and Power Allocation","date":"2023-02-04","arxiv_id":"2302.02250","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-learning-with-unsupervised-skill","title":"Developing Driving Strategies Efficiently: A Skill-Based Hierarchical Reinforcement Learning Approach","date":"2023-02-04","arxiv_id":"2302.02179","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-low-rank-mdps-with","title":"Reinforcement Learning in Low-Rank MDPs with Density Features","date":"2023-02-04","arxiv_id":"2302.02252","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-history-dependent","title":"Reinforcement Learning with History-Dependent Dynamic Contexts","date":"2023-02-04","arxiv_id":"2302.02061","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-cyber-system","title":"Deep Reinforcement Learning for Cyber System Defense under Dynamic Adversarial Uncertainties","date":"2023-02-03","arxiv_id":"2302.01595","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-online-error","title":"Deep Reinforcement Learning for Online Error Detection in Cyber-Physical Systems","date":"2023-02-03","arxiv_id":"2302.01567","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcing-user-retention-in-a-billion-scale","title":"Reinforcing User Retention in a Billion Scale Short Video Recommender System","date":"2023-02-03","arxiv_id":"2302.01724","repositories_listed":0,"syntology":null},{"url":null,"slug":"diversity-through-exclusion-dte-niche","title":"Diversity Through Exclusion (DTE): Niche Identification for Reinforcement Learning through Value-Decomposition","date":"2023-02-02","arxiv_id":"2302.01180","repositories_listed":0,"syntology":null},{"url":null,"slug":"performance-bounds-for-policy-based-average","title":"Performance Bounds for Policy-Based Average Reward Reinforcement Learning Algorithms","date":"2023-02-02","arxiv_id":"2302.01450","repositories_listed":0,"syntology":null},{"url":null,"slug":"reload-reinforcement-learning-with-optimistic","title":"ReLOAD: Reinforcement Learning with Optimistic Ascent-Descent for Last-Iterate Convergence in Constrained MDPs","date":"2023-02-02","arxiv_id":"2302.01275","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-physics-informed-neural-networks","title":"Bridging Physics-Informed Neural Networks with Reinforcement Learning: Hamilton-Jacobi-Bellman Proximal Policy Optimization (HJBPPO)","date":"2023-02-01","arxiv_id":"2302.00237","repositories_listed":0,"syntology":null},{"url":"/paper/collaborating-with-language-models-for","slug":"collaborating-with-language-models-for","title":"Collaborating with language models for embodied reasoning","date":"2023-02-01","arxiv_id":"2302.00763","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/collaborating-with-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2302.00763","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00763"}},"official":null}},{"url":null,"slug":"combining-tree-search-generative-models-and","title":"Combining Deep Reinforcement Learning and Search with Generative Models for Game-Theoretic Opponent Modeling","date":"2023-02-01","arxiv_id":"2302.00797","repositories_listed":0,"syntology":null},{"url":"/paper/efficient-multi-task-reinforcement-learning","slug":"efficient-multi-task-reinforcement-learning","title":"QMP: Q-switch Mixture of Policies for Multi-Task Behavior Sharing","date":"2023-02-01","arxiv_id":"2302.00671","repositories_listed":0,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-multi-task-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2302.00671","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.00671"}},"official":null}},{"url":null,"slug":"multi-zone-hvac-control-with-model-based-deep","title":"Multi-zone HVAC Control with Model-Based Deep Reinforcement Learning","date":"2023-02-01","arxiv_id":"2302.00725","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-fitted-q-evaluation-and-iteration","title":"Robust Fitted-Q-Evaluation and Iteration under Sequentially Exogenous Unobserved Confounders","date":"2023-02-01","arxiv_id":"2302.00662","repositories_listed":0,"syntology":null},{"url":null,"slug":"selective-uncertainty-propagation-in-offline","title":"Selective Uncertainty Propagation in Offline RL","date":"2023-02-01","arxiv_id":"2302.00284","repositories_listed":0,"syntology":null}],"record_sha256":"4c6223babf176b8eb9f96d19ad10701657cecf40f17a835fd142502054bf61a9","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}