{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/84","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":84,"pages_in_order":152,"rows_per_page":100,"rows":[8301,8400],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/83","next":"/task/reinforcement-learning-1/papers/85","papers":[{"url":null,"slug":"empathetic-persuasion-reinforcing-empathy-and-1","title":"Empathetic Persuasion: Reinforcing Empathy and Persuasiveness in Dialogue Systems","date":"2022-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"interactive-learning-from-natural-language","title":"Interactive Learning from Natural Language and Demonstrations using Signal Temporal Logic","date":"2022-07-01","arxiv_id":"2207.00627","repositories_listed":0,"syntology":null},{"url":null,"slug":"partner-personas-generation-for-dialogue","title":"Partner Personas Generation for Dialogue Response Generation","date":"2022-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-user-guided","title":"Reinforcement Learning Based User-Guided Motion Planning for Human-Robot Collaboration","date":"2022-07-01","arxiv_id":"2207.00492","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-of-multi-domain-dialog-1","title":"Reinforcement Learning of Multi-Domain Dialog Policies Via Action Embeddings","date":"2022-07-01","arxiv_id":"2207.00468","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-decision-making-for-lane-change-of","title":"Safe Decision-making for Lane-change of Autonomous Vehicles via Human Demonstration-aided Reinforcement Learning","date":"2022-07-01","arxiv_id":"2207.00448","repositories_listed":0,"syntology":null},{"url":null,"slug":"surf-semantic-level-unsupervised-reward","title":"SURF: Semantic-level Unsupervised Reward Function for Machine Translation","date":"2022-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"depth-cuprl-depth-imaged-contrastive","title":"Depth-CUPRL: Depth-Imaged Contrastive Unsupervised Prioritized Representations in Reinforcement Learning for Mapless Navigation of Unmanned Aerial Vehicles","date":"2022-06-30","arxiv_id":"2206.15211","repositories_listed":0,"syntology":null},{"url":null,"slug":"performative-reinforcement-learning","title":"Performative Reinforcement Learning","date":"2022-06-30","arxiv_id":"2207.00046","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-small-bowel","title":"Deep Reinforcement Learning for Small Bowel Path Tracking using Different Types of Annotations","date":"2022-06-29","arxiv_id":"2206.14847","repositories_listed":0,"syntology":null},{"url":null,"slug":"minimalist-and-high-performance","title":"Minimalist and High-performance Conversational Recommendation with Uncertainty Estimation for User Preference","date":"2022-06-29","arxiv_id":"2206.14468","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-reinforcement-learning-for-1","title":"Provably Efficient Reinforcement Learning for Online Adaptive Influence Maximization","date":"2022-06-29","arxiv_id":"2206.14846","repositories_listed":0,"syntology":null},{"url":null,"slug":"applications-of-reinforcement-learning-in-1","title":"Applications of Reinforcement Learning in Finance -- Trading with a Double Deep Q-Network","date":"2022-06-28","arxiv_id":"2206.14267","repositories_listed":0,"syntology":null},{"url":null,"slug":"dependency-parsing-with-backtracking-using","title":"Dependency Parsing with Backtracking using Deep Reinforcement Learning","date":"2022-06-28","arxiv_id":"2206.13914","repositories_listed":0,"syntology":null},{"url":null,"slug":"gan-based-intrinsic-exploration-for-sample","title":"GAN-based Intrinsic Exploration For Sample Efficient Reinforcement Learning","date":"2022-06-28","arxiv_id":"2206.14256","repositories_listed":0,"syntology":null},{"url":null,"slug":"masked-world-models-for-visual-control","title":"Masked World Models for Visual Control","date":"2022-06-28","arxiv_id":"2206.14244","repositories_listed":0,"syntology":null},{"url":null,"slug":"position-agnostic-autonomous-navigation-in","title":"Position-Agnostic Autonomous Navigation in Vineyards with Deep Reinforcement Learning","date":"2022-06-28","arxiv_id":"2206.14155","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-dynamic-model-1","title":"Reinforcement Learning Based Dynamic Model Combination for Time Series Forecasting","date":"2022-06-28","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-medical-image","title":"Reinforcement Learning in Medical Image Analysis: Concepts, Applications, Challenges, and Future Directions","date":"2022-06-28","arxiv_id":"2206.14302","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-perspective-exploration-in","title":"Risk Perspective Exploration in Distributional Reinforcement Learning","date":"2022-06-28","arxiv_id":"2206.14170","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatial-positioning-token-sptoken-for-smart-1","title":"Spatial Positioning Token (SPToken) for Smart Parking","date":"2022-06-28","arxiv_id":"2206.13880","repositories_listed":0,"syntology":null},{"url":null,"slug":"traffic-management-of-autonomous-vehicles","title":"Traffic Management of Autonomous Vehicles using Policy Based Deep Reinforcement Learning and Intelligent Routing","date":"2022-06-28","arxiv_id":"2206.14608","repositories_listed":0,"syntology":null},{"url":null,"slug":"emvlight-a-multi-agent-reinforcement-learning","title":"EMVLight: a Multi-agent Reinforcement Learning Framework for an Emergency Vehicle Decentralized Routing and Traffic Signal Control System","date":"2022-06-27","arxiv_id":"2206.13441","repositories_listed":0,"syntology":null},{"url":null,"slug":"humans-are-not-boltzmann-distributions","title":"Humans are not Boltzmann Distributions: Challenges and Opportunities for Modelling Human Feedback and Interaction in Reinforcement Learning","date":"2022-06-27","arxiv_id":"2206.13316","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-hidden-markov-model-based-deep","title":"Interpretable Hidden Markov Model-Based Deep Reinforcement Learning Hierarchical Framework for Predictive Maintenance of Turbofan Engines","date":"2022-06-27","arxiv_id":"2206.13433","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-complexity-of-adversarial-decision","title":"On the Complexity of Adversarial Decision Making","date":"2022-06-27","arxiv_id":"2206.13063","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-stochastic-processes-through","title":"Analysis of Stochastic Processes through Replay Buffers","date":"2022-06-26","arxiv_id":"2206.12848","repositories_listed":0,"syntology":null},{"url":null,"slug":"estimating-link-flows-in-road-networks-with","title":"Estimating Link Flows in Road Networks with Synthetic Trajectory Data Generation: Reinforcement Learning-based Approaches","date":"2022-06-26","arxiv_id":"2206.12873","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-the-need-for-blood-transfusion-in","title":"Predicting the Need for Blood Transfusion in Intensive Care Units with Reinforcement Learning","date":"2022-06-26","arxiv_id":"2206.14198","repositories_listed":0,"syntology":null},{"url":null,"slug":"functional-optimization-reinforcement","title":"Functional Optimization Reinforcement Learning for Real-Time Bidding","date":"2022-06-25","arxiv_id":"2206.13939","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-with-4","title":"Hierarchical Reinforcement Learning with Opponent Modeling for Distributed Multi-agent Cooperation","date":"2022-06-25","arxiv_id":"2206.12718","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-modern-card-games-with-large-scale","title":"Towards Modern Card Games with Large-Scale Action Spaces Through Action Representation","date":"2022-06-25","arxiv_id":"2206.12700","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-consistent-representation-learning-for","title":"Value-Consistent Representation Learning for Data-Efficient Reinforcement Learning","date":"2022-06-25","arxiv_id":"2206.12542","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-network-congestion-pricing-based-on","title":"Dynamic network congestion pricing based on deep reinforcement learning","date":"2022-06-24","arxiv_id":"2206.12188","repositories_listed":0,"syntology":null},{"url":null,"slug":"eco-driving-for-electric-connected-vehicles","title":"Eco-driving for Electric Connected Vehicles at Signalized Intersections: A Parameterized Reinforcement Learning approach","date":"2022-06-24","arxiv_id":"2206.12065","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-representation-training-in-sequential","title":"Joint Representation Training in Sequential Tasks with Shared Structure","date":"2022-06-24","arxiv_id":"2206.12441","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-the-policy-for-mixed-electric","title":"Learning the policy for mixed electric platoon control of automated and human-driven vehicles at signalized intersection: a random search approach","date":"2022-06-24","arxiv_id":"2206.12052","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-adaptive-platoon-and-reservation","title":"Modeling Adaptive Platoon and Reservation Based Autonomous Intersection Control: A Deep Reinforcement Learning Approach","date":"2022-06-24","arxiv_id":"2206.12419","repositories_listed":0,"syntology":null},{"url":null,"slug":"phasic-self-imitative-reduction-for-sparse","title":"Phasic Self-Imitative Reduction for Sparse-Reward Goal-Conditioned Reinforcement Learning","date":"2022-06-24","arxiv_id":"2206.12030","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-reinforcement-learning-in-1","title":"Provably Efficient Reinforcement Learning in Partially Observable Dynamical Systems","date":"2022-06-24","arxiv_id":"2206.12020","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-function-decomposition-for-iterative","title":"Value Function Decomposition for Iterative Design of Reinforcement Learning Agents","date":"2022-06-24","arxiv_id":"2206.13901","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-federated-reinforcement-learning-method","title":"A Federated Reinforcement Learning Method with Quantization for Cooperative Edge Caching in Fog Radio Access Networks","date":"2022-06-23","arxiv_id":"2206.11556","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-agile-skills-via-adversarial","title":"Learning Agile Skills via Adversarial Imitation of Rough Partial Demonstrations","date":"2022-06-23","arxiv_id":"2206.11693","repositories_listed":0,"syntology":null},{"url":null,"slug":"nearly-minimax-optimal-reinforcement-learning-1","title":"Nearly Minimax Optimal Reinforcement Learning with Linear Function Approximation","date":"2022-06-23","arxiv_id":"2206.11489","repositories_listed":0,"syntology":null},{"url":null,"slug":"recursive-reinforcement-learning","title":"Recursive Reinforcement Learning","date":"2022-06-23","arxiv_id":"2206.11430","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-under-partial","title":"Reinforcement Learning under Partial Observability Guided by Learned Environment Models","date":"2022-06-23","arxiv_id":"2206.11708","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-real-deal-a-review-of-challenges-and","title":"The Real Deal: A Review of Challenges and Opportunities in Moving Reinforcement Learning-Based Traffic Signal Control Systems Towards Reality","date":"2022-06-23","arxiv_id":"2206.11996","repositories_listed":0,"syntology":null},{"url":null,"slug":"auto-encoding-adversarial-imitation-learning","title":"Auto-Encoding Adversarial Imitation Learning","date":"2022-06-22","arxiv_id":"2206.11004","repositories_listed":0,"syntology":null},{"url":null,"slug":"curious-exploration-via-structured-world","title":"Curious Exploration via Structured World Models Yields Zero-Shot Object Manipulation","date":"2022-06-22","arxiv_id":"2206.11403","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-gossip-based-stochastic-bilevel","title":"Decentralized Gossip-Based Stochastic Bilevel Optimization over Communication Networks","date":"2022-06-22","arxiv_id":"2206.10870","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusion-of-model-free-reinforcement-learning","title":"Fusion of Model-free Reinforcement Learning with Microgrid Control: Review and Vision","date":"2022-06-22","arxiv_id":"2206.11398","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-optimal-treatment-strategies-for","title":"Learning Optimal Treatment Strategies for Sepsis Using Offline Reinforcement Learning in Continuous Space","date":"2022-06-22","arxiv_id":"2206.11190","repositories_listed":0,"syntology":null},{"url":null,"slug":"projection-free-constrained-stochastic","title":"Constrained Stochastic Nonconvex Optimization with State-dependent Markov Data","date":"2022-06-22","arxiv_id":"2206.11346","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-single-timescale-analysis-for-stochastic","title":"A Single-Timescale Analysis For Stochastic Approximation With Multiple Coupled Sequences","date":"2022-06-21","arxiv_id":"2206.10414","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-reinforcement-learning-linear","title":"Federated Stochastic Approximation under Markov Noise and Heterogeneity: Applications in Reinforcement Learning","date":"2022-06-21","arxiv_id":"2206.10185","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-optimal-policy-for-queueing-models","title":"Finding Optimal Policy for Queueing Models: New Parameterization","date":"2022-06-21","arxiv_id":"2206.10073","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybridization-of-evolutionary-algorithm-and","title":"Hybridization of evolutionary algorithm and deep reinforcement learning for multi-objective orienteering optimization","date":"2022-06-21","arxiv_id":"2206.10464","repositories_listed":0,"syntology":null},{"url":null,"slug":"imitate-then-transcend-multi-agent-optimal","title":"Imitate then Transcend: Multi-Agent Optimal Execution with Dual-Window Denoise PPO","date":"2022-06-21","arxiv_id":"2206.10736","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-voice-instructions-in-model","title":"Incorporating Voice Instructions in Model-Based Reinforcement Learning for Self-Driving Cars","date":"2022-06-21","arxiv_id":"2206.10249","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-imitation-learning-using-entropy","title":"Model-Based Imitation Learning Using Entropy Regularization of Model and Policy","date":"2022-06-21","arxiv_id":"2206.10101","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-statistical-efficiency-of-reward-free","title":"On the Statistical Efficiency of Reward-Free Exploration in Non-Linear RL","date":"2022-06-21","arxiv_id":"2206.10770","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-and-psychologically-pleasant-traffic","title":"Safe and Psychologically Pleasant Traffic Signal Control with Reinforcement Learning using Action Masking","date":"2022-06-21","arxiv_id":"2206.10122","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-integration-of-machine-learning-into","title":"The Integration of Machine Learning into Automated Test Generation: A Systematic Mapping Study","date":"2022-06-21","arxiv_id":"2206.10210","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-reinforcement-learning-for-1","title":"Constrained Reinforcement Learning for Robotics via Scenario-Based Programming","date":"2022-06-20","arxiv_id":"2206.09603","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforced-active-learning-for-multi","title":"Deep reinforced active learning for multi-class image classification","date":"2022-06-20","arxiv_id":"2206.13391","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-multi-agent-to-multi-robot-a-scalable","title":"From Multi-agent to Multi-robot: A Scalable Training and Evaluation Platform for Multi-robot Reinforcement Learning","date":"2022-06-20","arxiv_id":"2206.09590","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-safe-shooting-model-based","title":"Guided Safe Shooting: model based reinforcement learning with safety constraints","date":"2022-06-20","arxiv_id":"2206.09743","repositories_listed":0,"syntology":null},{"url":null,"slug":"s2rl-do-we-really-need-to-perceive-all-states","title":"S2RL: Do We Really Need to Perceive All States in Deep Multi-Agent Reinforcement Learning?","date":"2022-06-20","arxiv_id":"2206.11054","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-model-based-reinforcement","title":"A Survey on Model-based Reinforcement Learning","date":"2022-06-19","arxiv_id":"2206.09328","repositories_listed":0,"syntology":null},{"url":null,"slug":"guarantees-for-epsilon-greedy-reinforcement","title":"Guarantees for Epsilon-Greedy Reinforcement Learning with Function Approximation","date":"2022-06-19","arxiv_id":"2206.09421","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-multi-task-transferable-rewards-via","title":"Learning Multi-Task Transferable Rewards via Variational Inverse Reinforcement Learning","date":"2022-06-19","arxiv_id":"2206.09498","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-hop-age-of-information-scheduling-for","title":"Two-Hop Age of Information Scheduling for Multi-UAV Assisted Mobile Edge Computing: FRL vs MADDPG","date":"2022-06-19","arxiv_id":"2206.09488","repositories_listed":0,"syntology":null},{"url":null,"slug":"anymorph-learning-transferable-polices-by","title":"AnyMorph: Learning Transferable Polices By Inferring Agent Morphology","date":"2022-06-17","arxiv_id":"2206.12279","repositories_listed":0,"syntology":null},{"url":"/paper/bootstrapped-transformer-for-offline","slug":"bootstrapped-transformer-for-offline","title":"Bootstrapped Transformer for Offline Reinforcement Learning","date":"2022-06-17","arxiv_id":"2206.08569","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bootstrapped-transformer-for-offline#ran","syntology_url":"https://syntology.ai/paper/2206.08569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08569"}},"official":null}},{"url":null,"slug":"deep-reinforcement-learning-for-fmri","title":"Deep reinforcement learning for fMRI prediction of Autism Spectrum Disorder","date":"2022-06-17","arxiv_id":"2206.11224","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalised-policy-improvement-with-geometric","title":"Generalised Policy Improvement with Geometric Policy Composition","date":"2022-06-17","arxiv_id":"2206.08736","repositories_listed":0,"syntology":null},{"url":null,"slug":"backbones-review-feature-extraction-networks","title":"Backbones-Review: Feature Extraction Networks for Deep Learning and Deep Reinforcement Learning Approaches","date":"2022-06-16","arxiv_id":"2206.08016","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-macroeconomic","title":"Reinforcement Learning for Economic Policy: A New Frontier?","date":"2022-06-16","arxiv_id":"2206.08781","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-decision-time-vs-background","title":"A Look at Value-Based Decision-Time vs. Background Planning Methods Across Different Settings","date":"2022-06-16","arxiv_id":"2206.08442","repositories_listed":0,"syntology":null},{"url":null,"slug":"automating-the-resolution-of-flight-conflicts","title":"Automating the resolution of flight conflicts: Deep reinforcement learning in service of air traffic controllers","date":"2022-06-15","arxiv_id":"2206.07403","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-platoon-control-with-integrated","title":"Autonomous Platoon Control with Integrated Deep Reinforcement Learning and Dynamic Programming","date":"2022-06-15","arxiv_id":"2206.07536","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-learning-as-goal-conditioned","title":"Contrastive Learning as Goal-Conditioned Reinforcement Learning","date":"2022-06-15","arxiv_id":"2206.07568","repositories_listed":0,"syntology":null},{"url":null,"slug":"mean-semivariance-policy-optimization-via","title":"Mean-Semivariance Policy Optimization via Risk-Averse Reinforcement Learning","date":"2022-06-15","arxiv_id":"2206.07376","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-reinforcement-learning-for","title":"Rethinking Reinforcement Learning for Recommendation: A Prompt Perspective","date":"2022-06-15","arxiv_id":"2206.07353","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-some-common-practices-in","title":"Revisiting Some Common Practices in Cooperative Multi-Agent Reinforcement Learning","date":"2022-06-15","arxiv_id":"2206.07505","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-exact","title":"Deep Reinforcement Learning for Exact Combinatorial Optimization: Learning to Branch","date":"2022-06-14","arxiv_id":"2206.06965","repositories_listed":0,"syntology":null},{"url":null,"slug":"freekd-free-direction-knowledge-distillation","title":"FreeKD: Free-direction Knowledge Distillation for Graph Neural Networks","date":"2022-06-14","arxiv_id":"2206.06561","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-ended-learning-strategies-for-learning","title":"Open-Ended Learning Strategies for Learning Complex Locomotion Skills","date":"2022-06-14","arxiv_id":"2206.06796","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-with","title":"Robust Reinforcement Learning with Distributional Risk-averse formulation","date":"2022-06-14","arxiv_id":"2206.06841","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-the-capacitated-vehicle-routing","title":"Solving the capacitated vehicle routing problem with timing windows using rollouts and MAX-SAT","date":"2022-06-14","arxiv_id":"2206.06618","repositories_listed":0,"syntology":null},{"url":null,"slug":"stein-variational-goal-generation-for","title":"Stein Variational Goal Generation for adaptive Exploration in Multi-Goal Reinforcement Learning","date":"2022-06-14","arxiv_id":"2206.06719","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-solution-to-bongard-problems-a","title":"Towards a Solution to Bongard Problems: A Causal Approach","date":"2022-06-14","arxiv_id":"2206.07196","repositories_listed":0,"syntology":null},{"url":null,"slug":"variance-reduction-for-policy-gradient","title":"Variance Reduction for Policy-Gradient Methods via Empirical Variance Minimization","date":"2022-06-14","arxiv_id":"2206.06827","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-radial-basis-q-network","title":"Visual Radial Basis Q-Network","date":"2022-06-14","arxiv_id":"2206.06712","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-randomization-effects-on-sim2real","title":"Analysis of Randomization Effects on Sim2Real Transfer in Reinforcement Learning for Robotic Manipulation Tasks","date":"2022-06-13","arxiv_id":"2206.06282","repositories_listed":0,"syntology":null},{"url":null,"slug":"computation-offloading-and-resource-1","title":"Computation Offloading and Resource Allocation in F-RANs: A Federated Deep Reinforcement Learning Approach","date":"2022-06-13","arxiv_id":"2206.05881","repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsically-motivated-option-learning-a","title":"Intrinsically motivated option learning: a comparative study of recent methods","date":"2022-06-13","arxiv_id":"2206.06007","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-benefit-of-multitask-representation","title":"Provable Benefit of Multitask Representation Learning in Reinforcement Learning","date":"2022-06-13","arxiv_id":"2206.05900","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-offline-reinforcement","title":"Provably Efficient Offline Reinforcement Learning with Trajectory-Wise Reward","date":"2022-06-13","arxiv_id":"2206.06426","repositories_listed":0,"syntology":null},{"url":null,"slug":"relative-policy-transition-optimization-for","title":"Relative Policy-Transition Optimization for Fast Policy Transfer","date":"2022-06-13","arxiv_id":"2206.06009","repositories_listed":0,"syntology":null}],"record_sha256":"6953ed4fc01495c074c2ebb5aab32e7d8536d8dd557869ff44f1f525f58d9b94","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}