{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/79","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":79,"pages_in_order":135,"rows_per_page":100,"rows":[7801,7900],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/78","next":"/task/reinforcement-learning-2/papers/80","papers":[{"url":null,"slug":"learning-the-policy-for-mixed-electric","title":"Learning the policy for mixed electric platoon control of automated and human-driven vehicles at signalized intersection: a random search approach","date":"2022-06-24","arxiv_id":"2206.12052","repositories_listed":0,"syntology":null},{"url":null,"slug":"phasic-self-imitative-reduction-for-sparse","title":"Phasic Self-Imitative Reduction for Sparse-Reward Goal-Conditioned Reinforcement Learning","date":"2022-06-24","arxiv_id":"2206.12030","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-reinforcement-learning-in-1","title":"Provably Efficient Reinforcement Learning in Partially Observable Dynamical Systems","date":"2022-06-24","arxiv_id":"2206.12020","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-function-decomposition-for-iterative","title":"Value Function Decomposition for Iterative Design of Reinforcement Learning Agents","date":"2022-06-24","arxiv_id":"2206.13901","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-federated-reinforcement-learning-method","title":"A Federated Reinforcement Learning Method with Quantization for Cooperative Edge Caching in Fog Radio Access Networks","date":"2022-06-23","arxiv_id":"2206.11556","repositories_listed":0,"syntology":null},{"url":null,"slug":"nearly-minimax-optimal-reinforcement-learning-1","title":"Nearly Minimax Optimal Reinforcement Learning with Linear Function Approximation","date":"2022-06-23","arxiv_id":"2206.11489","repositories_listed":0,"syntology":null},{"url":null,"slug":"recursive-reinforcement-learning","title":"Recursive Reinforcement Learning","date":"2022-06-23","arxiv_id":"2206.11430","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-under-partial","title":"Reinforcement Learning under Partial Observability Guided by Learned Environment Models","date":"2022-06-23","arxiv_id":"2206.11708","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-gossip-based-stochastic-bilevel","title":"Decentralized Gossip-Based Stochastic Bilevel Optimization over Communication Networks","date":"2022-06-22","arxiv_id":"2206.10870","repositories_listed":0,"syntology":null},{"url":null,"slug":"fusion-of-model-free-reinforcement-learning","title":"Fusion of Model-free Reinforcement Learning with Microgrid Control: Review and Vision","date":"2022-06-22","arxiv_id":"2206.11398","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-optimal-treatment-strategies-for","title":"Learning Optimal Treatment Strategies for Sepsis Using Offline Reinforcement Learning in Continuous Space","date":"2022-06-22","arxiv_id":"2206.11190","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-reinforcement-learning-linear","title":"Federated Stochastic Approximation under Markov Noise and Heterogeneity: Applications in Reinforcement Learning","date":"2022-06-21","arxiv_id":"2206.10185","repositories_listed":0,"syntology":null},{"url":null,"slug":"finding-optimal-policy-for-queueing-models","title":"Finding Optimal Policy for Queueing Models: New Parameterization","date":"2022-06-21","arxiv_id":"2206.10073","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybridization-of-evolutionary-algorithm-and","title":"Hybridization of evolutionary algorithm and deep reinforcement learning for multi-objective orienteering optimization","date":"2022-06-21","arxiv_id":"2206.10464","repositories_listed":0,"syntology":null},{"url":null,"slug":"incorporating-voice-instructions-in-model","title":"Incorporating Voice Instructions in Model-Based Reinforcement Learning for Self-Driving Cars","date":"2022-06-21","arxiv_id":"2206.10249","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-imitation-learning-using-entropy","title":"Model-Based Imitation Learning Using Entropy Regularization of Model and Policy","date":"2022-06-21","arxiv_id":"2206.10101","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-and-psychologically-pleasant-traffic","title":"Safe and Psychologically Pleasant Traffic Signal Control with Reinforcement Learning using Action Masking","date":"2022-06-21","arxiv_id":"2206.10122","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-integration-of-machine-learning-into","title":"The Integration of Machine Learning into Automated Test Generation: A Systematic Mapping Study","date":"2022-06-21","arxiv_id":"2206.10210","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-reinforcement-learning-for-1","title":"Constrained Reinforcement Learning for Robotics via Scenario-Based Programming","date":"2022-06-20","arxiv_id":"2206.09603","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforced-active-learning-for-multi","title":"Deep reinforced active learning for multi-class image classification","date":"2022-06-20","arxiv_id":"2206.13391","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-multi-agent-to-multi-robot-a-scalable","title":"From Multi-agent to Multi-robot: A Scalable Training and Evaluation Platform for Multi-robot Reinforcement Learning","date":"2022-06-20","arxiv_id":"2206.09590","repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-safe-shooting-model-based","title":"Guided Safe Shooting: model based reinforcement learning with safety constraints","date":"2022-06-20","arxiv_id":"2206.09743","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-model-based-reinforcement","title":"A Survey on Model-based Reinforcement Learning","date":"2022-06-19","arxiv_id":"2206.09328","repositories_listed":0,"syntology":null},{"url":null,"slug":"guarantees-for-epsilon-greedy-reinforcement","title":"Guarantees for Epsilon-Greedy Reinforcement Learning with Function Approximation","date":"2022-06-19","arxiv_id":"2206.09421","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-multi-task-transferable-rewards-via","title":"Learning Multi-Task Transferable Rewards via Variational Inverse Reinforcement Learning","date":"2022-06-19","arxiv_id":"2206.09498","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-hop-age-of-information-scheduling-for","title":"Two-Hop Age of Information Scheduling for Multi-UAV Assisted Mobile Edge Computing: FRL vs MADDPG","date":"2022-06-19","arxiv_id":"2206.09488","repositories_listed":0,"syntology":null},{"url":null,"slug":"anymorph-learning-transferable-polices-by","title":"AnyMorph: Learning Transferable Polices By Inferring Agent Morphology","date":"2022-06-17","arxiv_id":"2206.12279","repositories_listed":0,"syntology":null},{"url":"/paper/bootstrapped-transformer-for-offline","slug":"bootstrapped-transformer-for-offline","title":"Bootstrapped Transformer for Offline Reinforcement Learning","date":"2022-06-17","arxiv_id":"2206.08569","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bootstrapped-transformer-for-offline#ran","syntology_url":"https://syntology.ai/paper/2206.08569","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08569"}},"official":null}},{"url":null,"slug":"deep-reinforcement-learning-for-fmri","title":"Deep reinforcement learning for fMRI prediction of Autism Spectrum Disorder","date":"2022-06-17","arxiv_id":"2206.11224","repositories_listed":0,"syntology":null},{"url":null,"slug":"backbones-review-feature-extraction-networks","title":"Backbones-Review: Feature Extraction Networks for Deep Learning and Deep Reinforcement Learning Approaches","date":"2022-06-16","arxiv_id":"2206.08016","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-in-macroeconomic","title":"Reinforcement Learning for Economic Policy: A New Frontier?","date":"2022-06-16","arxiv_id":"2206.08781","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-decision-time-vs-background","title":"A Look at Value-Based Decision-Time vs. Background Planning Methods Across Different Settings","date":"2022-06-16","arxiv_id":"2206.08442","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-platoon-control-with-integrated","title":"Autonomous Platoon Control with Integrated Deep Reinforcement Learning and Dynamic Programming","date":"2022-06-15","arxiv_id":"2206.07536","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-learning-as-goal-conditioned","title":"Contrastive Learning as Goal-Conditioned Reinforcement Learning","date":"2022-06-15","arxiv_id":"2206.07568","repositories_listed":0,"syntology":null},{"url":null,"slug":"mean-semivariance-policy-optimization-via","title":"Mean-Semivariance Policy Optimization via Risk-Averse Reinforcement Learning","date":"2022-06-15","arxiv_id":"2206.07376","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-reinforcement-learning-for","title":"Rethinking Reinforcement Learning for Recommendation: A Prompt Perspective","date":"2022-06-15","arxiv_id":"2206.07353","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-some-common-practices-in","title":"Revisiting Some Common Practices in Cooperative Multi-Agent Reinforcement Learning","date":"2022-06-15","arxiv_id":"2206.07505","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-exact","title":"Deep Reinforcement Learning for Exact Combinatorial Optimization: Learning to Branch","date":"2022-06-14","arxiv_id":"2206.06965","repositories_listed":0,"syntology":null},{"url":null,"slug":"freekd-free-direction-knowledge-distillation","title":"FreeKD: Free-direction Knowledge Distillation for Graph Neural Networks","date":"2022-06-14","arxiv_id":"2206.06561","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-reinforcement-learning-with","title":"Robust Reinforcement Learning with Distributional Risk-averse formulation","date":"2022-06-14","arxiv_id":"2206.06841","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-the-capacitated-vehicle-routing","title":"Solving the capacitated vehicle routing problem with timing windows using rollouts and MAX-SAT","date":"2022-06-14","arxiv_id":"2206.06618","repositories_listed":0,"syntology":null},{"url":null,"slug":"stein-variational-goal-generation-for","title":"Stein Variational Goal Generation for adaptive Exploration in Multi-Goal Reinforcement Learning","date":"2022-06-14","arxiv_id":"2206.06719","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-solution-to-bongard-problems-a","title":"Towards a Solution to Bongard Problems: A Causal Approach","date":"2022-06-14","arxiv_id":"2206.07196","repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsically-motivated-option-learning-a","title":"Intrinsically motivated option learning: a comparative study of recent methods","date":"2022-06-13","arxiv_id":"2206.06007","repositories_listed":0,"syntology":null},{"url":"/paper/object-tracking-using-siamese-network-based","slug":"object-tracking-using-siamese-network-based","title":"Object Tracking Using Siamese Network-Based Reinforcement Learning","date":"2022-06-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-benefit-of-multitask-representation","title":"Provable Benefit of Multitask Representation Learning in Reinforcement Learning","date":"2022-06-13","arxiv_id":"2206.05900","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-offline-reinforcement","title":"Provably Efficient Offline Reinforcement Learning with Trajectory-Wise Reward","date":"2022-06-13","arxiv_id":"2206.06426","repositories_listed":0,"syntology":null},{"url":null,"slug":"matching-options-to-tasks-using-option","title":"Matching options to tasks using Option-Indexed Hierarchical Reinforcement Learning","date":"2022-06-12","arxiv_id":"2206.05750","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-ea-a-reinforcement-learning-based","title":"RL-GA: A Reinforcement Learning-Based Genetic Algorithm for Electromagnetic Detection Satellite Scheduling Problem","date":"2022-06-12","arxiv_id":"2206.05694","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-offline-reinforcement-learning","title":"Federated Offline Reinforcement Learning","date":"2022-06-11","arxiv_id":"2206.05581","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-application-of-neural-networks-to-a","title":"An application of neural networks to a problem in knot theory and group theory (untangling braids)","date":"2022-06-10","arxiv_id":"2206.05373","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-multi-agent-reinforcement-learning-with-2","title":"Deep Multi-Agent Reinforcement Learning with Hybrid Action Spaces based on Maximum Entropy","date":"2022-06-10","arxiv_id":"2206.05108","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-mean-field-programming","title":"Dynamic mean field programming","date":"2022-06-10","arxiv_id":"2206.05200","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-retrieval-for-reinforcement","title":"Large-Scale Retrieval for Reinforcement Learning","date":"2022-06-10","arxiv_id":"2206.05314","repositories_listed":0,"syntology":null},{"url":null,"slug":"multifidelity-reinforcement-learning-with","title":"Multifidelity Reinforcement Learning with Control Variates","date":"2022-06-10","arxiv_id":"2206.05165","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-gradient-reinforcement-learning-for-1","title":"Policy Gradient Reinforcement Learning for Uncertain Polytopic LPV Systems based on MHE-MPC","date":"2022-06-10","arxiv_id":"2206.05089","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-optimization-method-assisted-ensemble-deep","title":"An Optimization Method-Assisted Ensemble Deep Reinforcement Learning Algorithm to Solve Unit Commitment Problems","date":"2022-06-09","arxiv_id":"2206.04249","repositories_listed":0,"syntology":null},{"url":null,"slug":"overcoming-the-spectral-bias-of-neural-value-1","title":"Overcoming the Spectral Bias of Neural Value Approximation","date":"2022-06-09","arxiv_id":"2206.04672","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-policy-iteration-via-amplitude","title":"Quantum Policy Iteration via Amplitude Estimation and Grover Search -- Towards Quantum Advantage for Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04741","repositories_listed":0,"syntology":null},{"url":null,"slug":"receding-horizon-inverse-reinforcement","title":"Receding Horizon Inverse Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04477","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-analysis-of-certainty-equivalence","title":"Regret Analysis of Certainty Equivalence Policies in Continuous-Time Linear-Quadratic Systems","date":"2022-06-09","arxiv_id":"2206.04434","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-bounds-for-information-directed","title":"Regret Bounds for Information-Directed Reinforcement Learning","date":"2022-06-09","arxiv_id":"2206.04640","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-in-2","title":"Sample-Efficient Reinforcement Learning in the Presence of Exogenous Information","date":"2022-06-09","arxiv_id":"2206.04282","repositories_listed":0,"syntology":null},{"url":null,"slug":"there-is-no-accuracy-interpretability","title":"There is no Accuracy-Interpretability Tradeoff in Reinforcement Learning for Mazes","date":"2022-06-09","arxiv_id":"2206.04266","repositories_listed":0,"syntology":null},{"url":null,"slug":"action-noise-in-off-policy-deep-reinforcement","title":"Action Noise in Off-Policy Deep Reinforcement Learning: Impact on Exploration and Performance","date":"2022-06-08","arxiv_id":"2206.03787","repositories_listed":0,"syntology":null},{"url":null,"slug":"few-shot-prompting-toward-controllable","title":"Learning to Generate Prompts for Dialogue Generation through Reinforcement Learning","date":"2022-06-08","arxiv_id":"2206.03931","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-is-minimax","title":"Model-Based Reinforcement Learning for Offline Zero-Sum Markov Games","date":"2022-06-08","arxiv_id":"2206.04044","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-inverse-scattering","title":"Reinforced Inverse Scattering","date":"2022-06-08","arxiv_id":"2206.04186","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-joint-learning-of-wireless-multiple","title":"Scalable Joint Learning of Wireless Multiple-Access Policies and their Signaling","date":"2022-06-08","arxiv_id":"2206.03844","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-online-disease-diagnosis-via-multi","title":"Scalable Online Disease Diagnosis via Multi-Model-Fused Actor-Critic Reinforcement Learning","date":"2022-06-08","arxiv_id":"2206.03659","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim2real-for-reinforcement-learning-driven","title":"Sim2real for Reinforcement Learning Driven Next Generation Networks","date":"2022-06-08","arxiv_id":"2206.03846","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-model-based-reinforcement-learning-approach-1","title":"A Model-Based Reinforcement Learning Approach for PID Design","date":"2022-06-07","arxiv_id":"2206.03567","repositories_listed":0,"syntology":null},{"url":null,"slug":"driving-in-real-life-with-inverse","title":"Driving in Real Life with Inverse Reinforcement Learning","date":"2022-06-07","arxiv_id":"2206.03004","repositories_listed":0,"syntology":null},{"url":null,"slug":"mix-mab-reinforcement-learning-based-resource","title":"MIX-MAB: Reinforcement Learning-based Resource Allocation Algorithm for LoRaWAN","date":"2022-06-07","arxiv_id":"2206.03401","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effectiveness-of-fine-tuning-versus","title":"On the Effectiveness of Fine-tuning Versus Meta-reinforcement Learning","date":"2022-06-07","arxiv_id":"2206.03271","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-role-of-discount-factor-in-offline","title":"On the Role of Discount Factor in Offline Reinforcement Learning","date":"2022-06-07","arxiv_id":"2206.03383","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-meta-reinforcement-learning-for","title":"Variational Meta Reinforcement Learning for Social Robotics","date":"2022-06-07","arxiv_id":"2206.03211","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-rollout-length-for-model-based-rl","title":"Adaptive Rollout Length for Model-Based RL Using Model-Free Deep RL","date":"2022-06-06","arxiv_id":"2206.02380","repositories_listed":0,"syntology":null},{"url":null,"slug":"asymptotic-instance-optimal-algorithms-for","title":"Asymptotic Instance-Optimal Algorithms for Interactive Decision Making","date":"2022-06-06","arxiv_id":"2206.02326","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-profit-risk-and-sustainability-for","title":"Balancing Profit, Risk, and Sustainability for Portfolio Management","date":"2022-06-06","arxiv_id":"2207.02134","repositories_listed":0,"syntology":null},{"url":null,"slug":"consensus-learning-for-cooperative-multi","title":"Consensus Learning for Cooperative Multi-Agent Reinforcement Learning","date":"2022-06-06","arxiv_id":"2206.02583","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-cybersecurity-1","title":"Deep Reinforcement Learning for Cybersecurity Threat Detection and Protection: A Review","date":"2022-06-06","arxiv_id":"2206.02733","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-entity-based-reinforcement-learning","title":"Efficient entity-based reinforcement learning","date":"2022-06-06","arxiv_id":"2206.02855","repositories_listed":0,"syntology":null},{"url":null,"slug":"real2sim-or-sim2real-robotics-visual","title":"Real2Sim or Sim2Real: Robotics Visual Insertion using Deep Reinforcement Learning and Real2Sim Policy Adaptation","date":"2022-06-06","arxiv_id":"2206.02679","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-sensitive-reinforcement-learning-1","title":"Provably Efficient Risk-Sensitive Reinforcement Learning: Iterated CVaR and Worst Path","date":"2022-06-06","arxiv_id":"2206.02678","repositories_listed":0,"syntology":null},{"url":null,"slug":"specification-guided-learning-of-nash","title":"Specification-Guided Learning of Nash Equilibria with High Social Welfare","date":"2022-06-06","arxiv_id":"2206.03348","repositories_listed":0,"syntology":null},{"url":null,"slug":"ddpg-based-on-multi-scale-strokes-for","title":"DDPG based on multi-scale strokes for financial time series trading strategy","date":"2022-06-05","arxiv_id":"2207.10071","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-dynamics-and-generalization-in","title":"Learning Dynamics and Generalization in Reinforcement Learning","date":"2022-06-05","arxiv_id":"2206.02126","repositories_listed":0,"syntology":null},{"url":null,"slug":"models-of-human-preference-for-learning","title":"Models of human preference for learning reward functions","date":"2022-06-05","arxiv_id":"2206.02231","repositories_listed":0,"syntology":null},{"url":null,"slug":"rapid-learning-of-spatial-representations-for","title":"Rapid Learning of Spatial Representations for Goal-Directed Navigation Based on a Novel Model of Hippocampal Place Fields","date":"2022-06-05","arxiv_id":"2206.02249","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-tree-backup-algorithms-for-temporal","title":"Adaptive Tree Backup Algorithms for Temporal-Difference Reinforcement Learning","date":"2022-06-04","arxiv_id":"2206.01896","repositories_listed":0,"syntology":null},{"url":null,"slug":"between-rate-distortion-theory-value","title":"Between Rate-Distortion Theory & Value Equivalence in Model-Based Reinforcement Learning","date":"2022-06-04","arxiv_id":"2206.02025","repositories_listed":0,"syntology":null},{"url":null,"slug":"deciding-what-to-model-value-equivalent","title":"Deciding What to Model: Value-Equivalent Sampling for Reinforcement Learning","date":"2022-06-04","arxiv_id":"2206.02072","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-value-estimation-for-off-policy","title":"Hybrid Value Estimation for Off-policy Evaluation and Offline Reinforcement Learning","date":"2022-06-04","arxiv_id":"2206.02000","repositories_listed":0,"syntology":null},{"url":null,"slug":"macc-cross-layer-multi-agent-congestion","title":"MACC: Cross-Layer Multi-Agent Congestion Control with Deep Reinforcement Learning","date":"2022-06-04","arxiv_id":"2206.01972","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-poisoning-attacks-on-offline-multi","title":"Reward Poisoning Attacks on Offline Multi-Agent Reinforcement Learning","date":"2022-06-04","arxiv_id":"2206.01888","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-framework-for-6","title":"A Deep Reinforcement Learning Framework For Column Generation","date":"2022-06-03","arxiv_id":"2206.02568","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentangling-epistemic-and-aleatoric","title":"Disentangling Epistemic and Aleatoric Uncertainty in Reinforcement Learning","date":"2022-06-03","arxiv_id":"2206.01558","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-energy-dispatch-and-unit-commitment-in","title":"Joint Energy Dispatch and Unit Commitment in Microgrids Based on Deep Reinforcement Learning","date":"2022-06-03","arxiv_id":"2206.01663","repositories_listed":0,"syntology":null},{"url":null,"slug":"kcrl-krasovskii-constrained-reinforcement","title":"KCRL: Krasovskii-Constrained Reinforcement Learning with Guaranteed Stability in Nonlinear Dynamical Systems","date":"2022-06-03","arxiv_id":"2206.01704","repositories_listed":0,"syntology":null}],"record_sha256":"91e38fb992401aeec5a382620b434a92a0c445b17f9dc89362e1ad7ab91475cb","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}