{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/65","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":65,"pages_in_order":152,"rows_per_page":100,"rows":[6401,6500],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/64","next":"/task/reinforcement-learning-1/papers/66","papers":[{"url":null,"slug":"discovering-command-and-control-c2-channels","title":"Discovering Command and Control (C2) Channels on Tor and Public Networks Using Reinforcement Learning","date":"2024-02-14","arxiv_id":"2402.09200","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-estimation-bias-in-deep-double-q","title":"Exploiting Estimation Bias in Clipped Double Q-Learning for Continous Control Reinforcement Learning Tasks","date":"2024-02-14","arxiv_id":"2402.09078","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-exploration-in-reinforcement","title":"How does Your RL Agent Explore? An Optimal Transport Analysis of Occupancy Measure Trajectories","date":"2024-02-14","arxiv_id":"2402.09113","repositories_listed":0,"syntology":null},{"url":null,"slug":"steady-state-error-compensation-for","title":"Steady-State Error Compensation for Reinforcement Learning with Quadratic Rewards","date":"2024-02-14","arxiv_id":"2402.09075","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-model-based-reinforcement","title":"Towards Robust Model-Based Reinforcement Learning Against Adversarial Corruption","date":"2024-02-14","arxiv_id":"2402.08991","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-agricultural-management","title":"Intelligent Agricultural Management Considering N$_2$O Emission and Climate Variability with Uncertainties","date":"2024-02-13","arxiv_id":"2402.08832","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-task-assignment-and-path-planning","title":"Optimal Task Assignment and Path Planning using Conflict-Based Search with Precedence and Temporal Constraints","date":"2024-02-13","arxiv_id":"2402.08772","repositories_listed":0,"syntology":null},{"url":null,"slug":"prdp-proximal-reward-difference-prediction","title":"PRDP: Proximal Reward Difference Prediction for Large-Scale Reward Finetuning of Diffusion Models","date":"2024-02-13","arxiv_id":"2402.08714","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-traffic-rule-compliance-in-safe","title":"Provable Traffic Rule Compliance in Safe Reinforcement Learning on the Open Sea","date":"2024-02-13","arxiv_id":"2402.08502","repositories_listed":0,"syntology":null},{"url":null,"slug":"auxiliary-reward-generation-with-transition","title":"Auxiliary Reward Generation with Transition Distance Representation Learning","date":"2024-02-12","arxiv_id":"2402.07412","repositories_listed":0,"syntology":null},{"url":null,"slug":"ir-aware-eco-timing-optimization-using","title":"IR-Aware ECO Timing Optimization Using Reinforcement Learning","date":"2024-02-12","arxiv_id":"2402.07781","repositories_listed":0,"syntology":null},{"url":null,"slug":"near-minimax-optimal-distributional","title":"Near-Minimax-Optimal Distributional Reinforcement Learning with a Generative Model","date":"2024-02-12","arxiv_id":"2402.07598","repositories_listed":0,"syntology":null},{"url":null,"slug":"future-prediction-can-be-a-strong-evidence-of","title":"Future Prediction Can be a Strong Evidence of Good History Representation in Partially Observable Environments","date":"2024-02-11","arxiv_id":"2402.07102","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-language-reinforcement-learning","title":"Natural Language Reinforcement Learning","date":"2024-02-11","arxiv_id":"2402.07157","repositories_listed":0,"syntology":null},{"url":null,"slug":"principled-penalty-based-methods-for-bilevel","title":"Principled Penalty-based Methods for Bilevel Reinforcement Learning and RLHF","date":"2024-02-10","arxiv_id":"2402.06886","repositories_listed":0,"syntology":null},{"url":null,"slug":"acter-diverse-and-actionable-counterfactual","title":"ACTER: Diverse and Actionable Counterfactual Sequences for Explaining and Diagnosing RL Policies","date":"2024-02-09","arxiv_id":"2402.06503","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-precision-geosteering-via-reinforcement","title":"High-Precision Geosteering via Reinforcement Learning and Particle Filters","date":"2024-02-09","arxiv_id":"2402.06377","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-to-teach-improve-sample-efficiency-in","title":"Learn to Teach: Sample-Efficient Privileged Learning for Humanoid Locomotion over Diverse Terrains","date":"2024-02-09","arxiv_id":"2402.06783","repositories_listed":0,"syntology":null},{"url":null,"slug":"rleegnet-integrating-brain-computer","title":"RLEEGNet: Integrating Brain-Computer Interfaces with Adaptive AI for Intuitive Responsiveness and High-Accuracy Motor Imagery Classification","date":"2024-02-09","arxiv_id":"2402.09465","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-function-interference-and-greedy-action","title":"Value function interference and greedy action selection in value-based multi-objective reinforcement learning","date":"2024-02-09","arxiv_id":"2402.06266","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentially-private-model-based-offline","title":"Differentially Private Deep Model-Based Reinforcement Learning","date":"2024-02-08","arxiv_id":"2402.05525","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-offline-reinforcement-learning-1","title":"Federated Offline Reinforcement Learning: Collaborative Single-Policy Coverage Suffices","date":"2024-02-08","arxiv_id":"2402.05876","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-fluid-directed-rigid-body-control","title":"Real-World Fluid Directed Rigid Body Control via Deep Reinforcement Learning","date":"2024-02-08","arxiv_id":"2402.06102","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-intelligent-agents-in-combat","title":"Scaling Intelligent Agents in Combat Simulations for Wargaming","date":"2024-02-08","arxiv_id":"2402.06694","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bayesian-approach-to-online-learning-for","title":"Context in Public Health for Underserved Communities: A Bayesian Approach to Online Restless Bandits","date":"2024-02-07","arxiv_id":"2402.04933","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-primal-dual-algorithm-for-offline","title":"A Primal-Dual Algorithm for Offline Constrained Reinforcement Learning with Linear MDPs","date":"2024-02-07","arxiv_id":"2402.04493","repositories_listed":0,"syntology":null},{"url":null,"slug":"code-as-reward-empowering-reinforcement","title":"Code as Reward: Empowering Reinforcement Learning with VLMs","date":"2024-02-07","arxiv_id":"2402.04764","repositories_listed":0,"syntology":null},{"url":null,"slug":"convergence-for-natural-policy-gradient-on","title":"Convergence for Natural Policy Gradient on Infinite-State Queueing MDPs","date":"2024-02-07","arxiv_id":"2402.05274","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-by-doing-an-online-causal","title":"Learning by Doing: An Online Causal Reinforcement Learning Framework with Causal-Aware Policy","date":"2024-02-07","arxiv_id":"2402.04869","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-diverse-policies-with-soft-self","title":"Learning Diverse Policies with Soft Self-Generated Guidance","date":"2024-02-07","arxiv_id":"2402.04539","repositories_listed":0,"syntology":null},{"url":"/paper/compound-returns-reduce-variance-in","slug":"compound-returns-reduce-variance-in","title":"Averaging $n$-step Returns Reduces Variance in Reinforcement Learning","date":"2024-02-06","arxiv_id":"2402.03903","repositories_listed":0,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/compound-returns-reduce-variance-in#ran","syntology_url":"https://syntology.ai/paper/2402.03903","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03903"}},"official":null}},{"url":null,"slug":"no-regret-reinforcement-learning-in-smooth","title":"No-Regret Reinforcement Learning in Smooth MDPs","date":"2024-02-06","arxiv_id":"2402.03792","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-from-bagged-reward-a","title":"Reinforcement Learning from Bagged Reward","date":"2024-02-06","arxiv_id":"2402.03771","repositories_listed":0,"syntology":null},{"url":null,"slug":"abstracted-trajectory-visualization-for","title":"Abstracted Trajectory Visualization for Explainability in Reinforcement Learning","date":"2024-02-05","arxiv_id":"2402.07928","repositories_listed":0,"syntology":null},{"url":null,"slug":"assessing-the-impact-of-distribution-shift-on","title":"Assessing the Impact of Distribution Shift on Reinforcement Learning Performance","date":"2024-02-05","arxiv_id":"2402.03590","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-diffuser-planning-towards-high","title":"Contrastive Diffuser: Planning Towards High Return States via Contrastive Learning","date":"2024-02-05","arxiv_id":"2402.02772","repositories_listed":0,"syntology":null},{"url":null,"slug":"frugal-actor-critic-sample-efficient-off","title":"Frugal Actor-Critic: Sample Efficient Off-Policy Deep Reinforcement Learning Using Unique Experiences","date":"2024-02-05","arxiv_id":"2402.05963","repositories_listed":0,"syntology":null},{"url":null,"slug":"iced-zero-shot-transfer-in-reinforcement","title":"DRED: Zero-Shot Transfer in Reinforcement Learning via Data-Regularised Environment Design","date":"2024-02-05","arxiv_id":"2402.03479","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-what-affects-generalization-gap","title":"Understanding What Affects the Generalization Gap in Visual Reinforcement Learning: Theory and Empirical Evidence","date":"2024-02-05","arxiv_id":"2402.02701","repositories_listed":0,"syntology":null},{"url":null,"slug":"utility-based-reinforcement-learning-unifying","title":"Utility-Based Reinforcement Learning: Unifying Single-objective and Multi-objective Reinforcement Learning","date":"2024-02-05","arxiv_id":"2402.02665","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-language-models-provide-promptable","title":"Vision-Language Models Provide Promptable Representations for Reinforcement Learning","date":"2024-02-05","arxiv_id":"2402.02651","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-safe-reinforcement-learning-driven-weights","title":"A Safe Reinforcement Learning driven Weights-varying Model Predictive Control for Autonomous Vehicle Motion Control","date":"2024-02-04","arxiv_id":"2402.02624","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffstitch-boosting-offline-reinforcement","title":"DiffStitch: Boosting Offline Reinforcement Learning with Diffusion-based Trajectory Stitching","date":"2024-02-04","arxiv_id":"2402.02439","repositories_listed":0,"syntology":null},{"url":null,"slug":"evading-deep-learning-based-malware-detectors","title":"Evading Deep Learning-Based Malware Detectors via Obfuscation: A Deep Reinforcement Learning Approach","date":"2024-02-04","arxiv_id":"2402.02600","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-virtues-of-pessimism-in-inverse","title":"The Virtues of Pessimism in Inverse Reinforcement Learning","date":"2024-02-04","arxiv_id":"2402.02616","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-constraint-formulations-in-safe","title":"A Survey of Constraint Formulations in Safe Reinforcement Learning","date":"2024-02-03","arxiv_id":"2402.02025","repositories_listed":0,"syntology":null},{"url":"/paper/value-aided-conditional-supervised-learning","slug":"value-aided-conditional-supervised-learning","title":"Adaptive $Q$-Aid for Conditional Supervised Learning in Offline Reinforcement Learning","date":"2024-02-03","arxiv_id":"2402.02017","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/value-aided-conditional-supervised-learning#ran","syntology_url":"https://syntology.ai/paper/2402.02017","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02017"}},"official":null}},{"url":null,"slug":"an-auction-based-marketplace-for-model","title":"An Auction-based Marketplace for Model Trading in Federated Learning","date":"2024-02-02","arxiv_id":"2402.01802","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-reinforcement-learning-for-routing","title":"Efficient Reinforcement Learning for Routing Jobs in Heterogeneous Queueing Systems","date":"2024-02-02","arxiv_id":"2402.01147","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-political-preferences-of-llms","title":"The Political Preferences of LLMs","date":"2024-02-02","arxiv_id":"2402.01789","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-rl-llm-taxonomy-tree-reviewing-synergies","title":"The RL/LLM Taxonomy Tree: Reviewing Synergies Between Reinforcement Learning and Large Language Models","date":"2024-02-02","arxiv_id":"2402.01874","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-based-controller-to","title":"A Reinforcement Learning Based Controller to Minimize Forces on the Crutches of a Lower-Limb Exoskeleton","date":"2024-01-31","arxiv_id":"2402.00135","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-graph-for-multi-robot-social","title":"Attention Graph for Multi-Robot Social Navigation with Deep Reinforcement Learning","date":"2024-01-31","arxiv_id":"2401.17914","repositories_listed":0,"syntology":null},{"url":null,"slug":"causal-coordinated-concurrent-reinforcement","title":"Causal Coordinated Concurrent Reinforcement Learning","date":"2024-01-31","arxiv_id":"2401.18012","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-based-eco-driving","title":"Safe Reinforcement Learning-Based Eco-Driving Control for Mixed Traffic Flows With Disturbances","date":"2024-01-31","arxiv_id":"2401.17837","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-versatile-dynamic","title":"Reinforcement Learning for Versatile, Dynamic, and Robust Bipedal Locomotion Control","date":"2024-01-30","arxiv_id":"2401.16889","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-former-stitching-via-latent","title":"Context-Former: Stitching via Latent Conditioned Sequence Modeling","date":"2024-01-29","arxiv_id":"2401.16452","repositories_listed":0,"syntology":null},{"url":null,"slug":"look-around-unexpected-gains-from-training-on","title":"The Indoor-Training Effect: unexpected gains from distribution shifts in the transition function","date":"2024-01-29","arxiv_id":"2401.15856","repositories_listed":0,"syntology":null},{"url":null,"slug":"serl-a-software-suite-for-sample-efficient","title":"SERL: A Software Suite for Sample-Efficient Robotic Reinforcement Learning","date":"2024-01-29","arxiv_id":"2401.16013","repositories_listed":0,"syntology":null},{"url":null,"slug":"social-interpretable-reinforcement-learning","title":"Social Interpretable Reinforcement Learning","date":"2024-01-27","arxiv_id":"2401.15480","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-limitations-of-markovian-rewards-to","title":"On the Limitations of Markovian Rewards to Express Multi-Objective, Risk-Sensitive, and Modal Tasks","date":"2024-01-26","arxiv_id":"2401.14811","repositories_listed":0,"syntology":null},{"url":null,"slug":"constant-stepsize-q-learning-distributional","title":"Constant Stepsize Q-learning: Distributional Convergence, Bias and Extrapolation","date":"2024-01-25","arxiv_id":"2401.13884","repositories_listed":0,"syntology":null},{"url":null,"slug":"hi-core-hierarchical-knowledge-transfer-for","title":"Hierarchical Continual Reinforcement Learning via Large Language Model","date":"2024-01-25","arxiv_id":"2401.15098","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-sensing-and-computing-decision","title":"Learning-based sensing and computing decision for data freshness in edge computing-enabled networks","date":"2024-01-25","arxiv_id":"2401.13936","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-fast-changing-slow-in-spiking-neural","title":"Learning fast changing slow in spiking neural networks","date":"2024-01-25","arxiv_id":"2402.10069","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-reinforcement-learning-by","title":"Sample Efficient Reinforcement Learning by Automatically Learning to Compose Subtasks","date":"2024-01-25","arxiv_id":"2401.14226","repositories_listed":0,"syntology":null},{"url":null,"slug":"scilab-rl-a-software-framework-for-efficient","title":"Scilab-RL: A software framework for efficient reinforcement learning and cognitive modeling research","date":"2024-01-25","arxiv_id":"2401.14488","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-safe-reinforcement-learning-algorithm-for","title":"A Safe Reinforcement Learning Algorithm for Supervisory Control of Power Plants","date":"2024-01-23","arxiv_id":"2401.13020","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-inference-as-a-model-of-agency","title":"Active Inference as a Model of Agency","date":"2024-01-23","arxiv_id":"2401.12917","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-minimal-and-reusable-causal-state","title":"Building Minimal and Reusable Causal State Abstractions for Reinforcement Learning","date":"2024-01-23","arxiv_id":"2401.12497","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-safety-critics-via-a-non-contractive","title":"Learning safety critics via a non-contractive binary bellman operator","date":"2024-01-23","arxiv_id":"2401.12849","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-stochastic-variance-reduced-proximal","title":"On the Stochastic (Variance-Reduced) Proximal Gradient Method for Regularized Expected Reward Optimization","date":"2024-01-23","arxiv_id":"2401.12508","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-socially-and-morally-aware-rl-agent","title":"Towards Socially and Morally Aware RL agent: Reward Design With LLM","date":"2024-01-23","arxiv_id":"2401.12459","repositories_listed":0,"syntology":null},{"url":null,"slug":"back-stepping-experience-replay-with","title":"Back-stepping Experience Replay with Application to Model-free Reinforcement Learning for a Soft Snake Robot","date":"2024-01-21","arxiv_id":"2401.11372","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-reinforcement-learning-for-5","title":"Constrained Reinforcement Learning for Adaptive Controller Synchronization in Distributed SDN","date":"2024-01-21","arxiv_id":"2403.08775","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-reinforcement-learning-for","title":"Large-scale Reinforcement Learning for Diffusion Models","date":"2024-01-20","arxiv_id":"2401.12244","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-efficient-generalizable-framework-for","title":"Efficient Training of Generalizable Visuomotor Policies via Control-Aware Augmentation","date":"2024-01-17","arxiv_id":"2401.09258","repositories_listed":0,"syntology":null},{"url":null,"slug":"crowd-prefrl-preference-based-reward-learning","title":"Crowd-PrefRL: Preference-Based Reward Learning from Crowds","date":"2024-01-17","arxiv_id":"2401.10941","repositories_listed":0,"syntology":null},{"url":null,"slug":"swbt-similarity-weighted-behavior-transformer","title":"Learning from Imperfect Demonstrations with Self-Supervision for Robotic Manipulation","date":"2024-01-17","arxiv_id":"2401.08957","repositories_listed":0,"syntology":null},{"url":null,"slug":"cyclight-learning-traffic-signal-cooperation","title":"CycLight: learning traffic signal cooperation with a cycle-level strategy","date":"2024-01-16","arxiv_id":"2401.08121","repositories_listed":0,"syntology":null},{"url":null,"slug":"prewrite-prompt-rewriting-with-reinforcement","title":"PRewrite: Prompt Rewriting with Reinforcement Learning","date":"2024-01-16","arxiv_id":"2401.08189","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-reinforcement-learning-with-free-form","title":"Safe Reinforcement Learning with Free-form Natural Language Constraints and Pre-Trained Language Models","date":"2024-01-15","arxiv_id":"2401.07553","repositories_listed":0,"syntology":null},{"url":null,"slug":"drlc-reinforcement-learning-with-dense","title":"Beyond Sparse Rewards: Enhancing Reinforcement Learning with Language Model Critique in Text Generation","date":"2024-01-14","arxiv_id":"2401.07382","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-from-llm-feedback-to","title":"Reinforcement Learning from LLM Feedback to Counteract Goal Misgeneralization","date":"2024-01-14","arxiv_id":"2401.07181","repositories_listed":0,"syntology":null},{"url":null,"slug":"bp-l-online-learning-via-synthetic-gradients","title":"BP(λ): Online Learning via Synthetic Gradients","date":"2024-01-13","arxiv_id":"2401.07044","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-command-and-control-channels","title":"Discovering Command and Control Channels Using Reinforcement Learning","date":"2024-01-13","arxiv_id":"2401.07154","repositories_listed":0,"syntology":null},{"url":null,"slug":"mutual-enhancement-of-large-language-and","title":"Mutual Enhancement of Large Language and Reinforcement Learning Models through Bi-Directional Feedback Mechanisms: A Case Study","date":"2024-01-12","arxiv_id":"2401.06603","repositories_listed":0,"syntology":null},{"url":null,"slug":"unex-rl-reinforcing-long-term-rewards-in","title":"UNEX-RL: Reinforcing Long-Term Rewards in Multi-Stage Recommender Systems with UNidirectional EXecution","date":"2024-01-12","arxiv_id":"2401.06470","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-reinforcement-learning-for-4","title":"Model-Free Reinforcement Learning for Automated Fluid Administration in Critical Care","date":"2024-01-11","arxiv_id":"2401.06299","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimistic-model-rollouts-for-pessimistic","title":"Optimistic Model Rollouts for Pessimistic Offline Policy Optimization","date":"2024-01-11","arxiv_id":"2401.05899","repositories_listed":0,"syntology":null},{"url":null,"slug":"parrot-pareto-optimal-multi-reward","title":"Parrot: Pareto-optimal Multi-Reward Reinforcement Learning Framework for Text-to-Image Generation","date":"2024-01-11","arxiv_id":"2401.05675","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-information-theoretic-approach-to-12","title":"An Information Theoretic Approach to Interaction-Grounded Learning","date":"2024-01-10","arxiv_id":"2401.05015","repositories_listed":0,"syntology":null},{"url":null,"slug":"innate-values-driven-reinforcement-learning","title":"Innate-Values-driven Reinforcement Learning based Cooperative Multi-Agent Cognitive Modeling","date":"2024-01-10","arxiv_id":"2401.05572","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-optimizing-rag-for","title":"Reinforcement Learning for Optimizing RAG for Domain Chatbots","date":"2024-01-10","arxiv_id":"2401.06800","repositories_listed":0,"syntology":null},{"url":null,"slug":"taming-data-hungry-reinforcement-learning","title":"Taming \"data-hungry\" reinforcement learning? Stability in continuous state-action spaces","date":"2024-01-10","arxiv_id":"2401.05233","repositories_listed":0,"syntology":null},{"url":null,"slug":"starcraftimage-a-dataset-for-prototyping-1","title":"StarCraftImage: A Dataset For Prototyping Spatial Reasoning Methods For Multi-Agent Environments","date":"2024-01-09","arxiv_id":"2401.04290","repositories_listed":0,"syntology":null},{"url":null,"slug":"behavioural-cloning-in-vizdoom","title":"Behavioural Cloning in VizDoom","date":"2024-01-08","arxiv_id":"2401.03993","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-multi-truck","title":"Deep Reinforcement Learning for Multi-Truck Vehicle Routing Problems with Multi-Leg Demand Routes","date":"2024-01-08","arxiv_id":"2401.08669","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-term-safe-reinforcement-learning-with","title":"Long-term Safe Reinforcement Learning with Binary Feedback","date":"2024-01-08","arxiv_id":"2401.03786","repositories_listed":0,"syntology":null},{"url":null,"slug":"novelgym-a-flexible-ecosystem-for-hybrid","title":"NovelGym: A Flexible Ecosystem for Hybrid Planning and Learning Agents Designed for Open Worlds","date":"2024-01-07","arxiv_id":"2401.03546","repositories_listed":0,"syntology":null}],"record_sha256":"9eec94603348559ecb672b908f9b3bc4b4d3c1e2d7b7f1a8d876ddde3f46285b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}