{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/96","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":96,"pages_in_order":135,"rows_per_page":100,"rows":[9501,9600],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/95","next":"/task/reinforcement-learning-2/papers/97","papers":[{"url":null,"slug":"decentralized-swarm-collision-avoidance-for","title":"Nearest-Neighbor-based Collision Avoidance for Quadrotors via Reinforcement Learning","date":"2021-04-30","arxiv_id":"2104.14912","repositories_listed":0,"syntology":null},{"url":null,"slug":"discrete-time-mean-field-control-with","title":"Discrete-Time Mean Field Control with Environment States","date":"2021-04-30","arxiv_id":"2104.14900","repositories_listed":0,"syntology":null},{"url":null,"slug":"mitigating-political-bias-in-language-models","title":"Mitigating Political Bias in Language Models Through Reinforced Calibration","date":"2021-04-30","arxiv_id":"2104.14795","repositories_listed":0,"syntology":null},{"url":null,"slug":"antagonistic-crowd-simulation-model","title":"Emotional Contagion-Aware Deep Reinforcement Learning for Antagonistic Crowd Simulation","date":"2021-04-29","arxiv_id":"2105.00854","repositories_listed":0,"syntology":null},{"url":null,"slug":"hypernetwork-dismantling-via-deep","title":"Hypernetwork Dismantling via Deep Reinforcement Learning","date":"2021-04-29","arxiv_id":"2104.14332","repositories_listed":0,"syntology":null},{"url":null,"slug":"maximum-entropy-inverse-reinforcement","title":"Adversarial Inverse Reinforcement Learning for Mean Field Games","date":"2021-04-29","arxiv_id":"2104.14654","repositories_listed":0,"syntology":null},{"url":null,"slug":"medium-access-using-distributed-reinforcement","title":"Medium Access using Distributed Reinforcement Learning for IoTs with Low-Complexity Wireless Transceivers","date":"2021-04-29","arxiv_id":"2104.14549","repositories_listed":0,"syntology":null},{"url":null,"slug":"pre-training-of-deep-rl-agents-for-improved","title":"Pre-training of Deep RL Agents for Improved Learning under Domain Randomization","date":"2021-04-29","arxiv_id":"2104.14386","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-meta-reinforcement-learning-to-bridge","title":"Using Meta Reinforcement Learning to Bridge the Gap between Simulation and Experiment in Energy Demand Response","date":"2021-04-29","arxiv_id":"2104.14670","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-is-going-on-inside-recurrent-meta","title":"What is Going on Inside Recurrent Meta Reinforcement Learning Agents?","date":"2021-04-29","arxiv_id":"2104.14644","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-generalized-projected-bellman-error-for-off","title":"A Generalized Projected Bellman Error for Off-policy Value Estimation in Reinforcement Learning","date":"2021-04-28","arxiv_id":"2104.13844","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-environment-for","title":"A Reinforcement Learning Environment for Polyhedral Optimizations","date":"2021-04-28","arxiv_id":"2104.13732","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-intersection-handling-using-multi","title":"End-to-End Intersection Handling using Multi-Agent Deep Reinforcement Learning","date":"2021-04-28","arxiv_id":"2104.13617","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-mis-design-for-autonomous-driving","title":"Reward (Mis)design for Autonomous Driving","date":"2021-04-28","arxiv_id":"2104.13906","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-adversarial-training-for-meta","title":"Adaptive Adversarial Training for Meta Reinforcement Learning","date":"2021-04-27","arxiv_id":"2104.13302","repositories_listed":0,"syntology":null},{"url":null,"slug":"controlling-earthquake-like-instabilities","title":"Controlling earthquake-like instabilities using artificial intelligence","date":"2021-04-27","arxiv_id":"2104.13180","repositories_listed":0,"syntology":null},{"url":null,"slug":"implementing-reinforcement-learning","title":"Implementing Reinforcement Learning Algorithms in Retail Supply Chains with OpenAI Gym Toolkit","date":"2021-04-27","arxiv_id":"2104.14398","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for-5","title":"A Deep Reinforcement Learning Approach for the Meal Delivery Problem","date":"2021-04-24","arxiv_id":"2104.12000","repositories_listed":0,"syntology":null},{"url":null,"slug":"disco-rl-distribution-conditioned","title":"DisCo RL: Distribution-Conditioned Reinforcement Learning for General-Purpose Policies","date":"2021-04-23","arxiv_id":"2104.11707","repositories_listed":0,"syntology":null},{"url":null,"slug":"formula-rl-deep-reinforcement-learning-for","title":"Formula RL: Deep Reinforcement Learning for Autonomous Racing using Telemetry Data","date":"2021-04-22","arxiv_id":"2104.11106","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-using-guided","title":"Reinforcement Learning using Guided Observability","date":"2021-04-22","arxiv_id":"2104.10986","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-fusion-for-adaptive-and-customizable","title":"Policy Fusion for Adaptive and Customizable Reinforcement Learning Agents","date":"2021-04-21","arxiv_id":"2104.10610","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-traffic-signal","title":"Reinforcement Learning for Traffic Signal Control: Comparison with Commercial Systems","date":"2021-04-21","arxiv_id":"2104.10455","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-an-aid-policy-to-minimize-student","title":"Discovering an Aid Policy to Minimize Student Evasion Using Offline Reinforcement Learning","date":"2021-04-20","arxiv_id":"2104.10258","repositories_listed":0,"syntology":null},{"url":null,"slug":"drl-deep-reinforcement-learning-for","title":"DRL: Deep Reinforcement Learning for Intelligent Robot Control -- Concept, Literature, and Future","date":"2021-04-20","arxiv_id":"2105.13806","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-predictive-control-and-reinforcement","title":"Model-predictive control and reinforcement learning in multi-energy system case studies","date":"2021-04-20","arxiv_id":"2104.09785","repositories_listed":0,"syntology":null},{"url":null,"slug":"network-defense-is-not-a-game","title":"Network Defense is Not a Game","date":"2021-04-20","arxiv_id":"2104.10262","repositories_listed":0,"syntology":null},{"url":null,"slug":"network-wide-traffic-signal-control","title":"Network-wide traffic signal control optimization using a multi-agent deep reinforcement learning","date":"2021-04-20","arxiv_id":"2104.09936","repositories_listed":0,"syntology":null},{"url":null,"slug":"outcome-driven-reinforcement-learning-via","title":"Outcome-Driven Reinforcement Learning via Variational Inference","date":"2021-04-20","arxiv_id":"2104.10190","repositories_listed":0,"syntology":null},{"url":null,"slug":"prospective-artificial-intelligence","title":"Prospective Artificial Intelligence Approaches for Active Cyber Defence","date":"2021-04-20","arxiv_id":"2104.09981","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-synthesis-of-verified-controllers-in","title":"Scalable Synthesis of Verified Controllers in Deep Reinforcement Learning","date":"2021-04-20","arxiv_id":"2104.10219","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-learning-for-financial-markets","title":"Adaptive learning for financial markets mixing model-based and model-free RL for volatility targeting","date":"2021-04-19","arxiv_id":"2104.10483","repositories_listed":0,"syntology":null},{"url":null,"slug":"agent-centric-representations-for-multi-agent","title":"Agent-Centric Representations for Multi-Agent Reinforcement Learning","date":"2021-04-19","arxiv_id":"2104.09402","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-multi-agent-fitted-q-iteration","title":"Approximated Multi-Agent Fitted Q Iteration","date":"2021-04-19","arxiv_id":"2104.09343","repositories_listed":0,"syntology":null},{"url":null,"slug":"constraints-satisfiability-driven","title":"Constraints Satisfiability Driven Reinforcement Learning for Autonomous Cyber Defense","date":"2021-04-19","arxiv_id":"2104.08994","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-in-a-monetary","title":"Deep Reinforcement Learning in a Monetary Model","date":"2021-04-19","arxiv_id":"2104.09368","repositories_listed":0,"syntology":null},{"url":null,"slug":"singular-perturbation-based-reinforcement","title":"Singular Perturbation-based Reinforcement Learning of Two-Point Boundary Optimal Control Systems","date":"2021-04-19","arxiv_id":"2104.09652","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-value-aligned-reinforcement-learning","title":"Training Value-Aligned Reinforcement Learning Agents Using a Normative Prior","date":"2021-04-19","arxiv_id":"2104.09469","repositories_listed":0,"syntology":null},{"url":null,"slug":"mt-opt-continuous-multi-task-robotic","title":"MT-Opt: Continuous Multi-Task Robotic Reinforcement Learning at Scale","date":"2021-04-16","arxiv_id":"2104.08212","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-exploration-in-model-based-reinforcement","title":"Safe Exploration in Model-based Reinforcement Learning using Control Barrier Functions","date":"2021-04-16","arxiv_id":"2104.08171","repositories_listed":0,"syntology":null},{"url":null,"slug":"actionable-models-unsupervised-offline","title":"Actionable Models: Unsupervised Offline Reinforcement Learning of Robotic Skills","date":"2021-04-15","arxiv_id":"2104.07749","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-based-3","title":"Multi-Agent Reinforcement Learning Based Coded Computation for Mobile Ad Hoc Computing","date":"2021-04-15","arxiv_id":"2104.07539","repositories_listed":0,"syntology":null},{"url":null,"slug":"rule-based-reinforcement-learning-for","title":"Rule-Based Reinforcement Learning for Efficient Robot Navigation with Space Reduction","date":"2021-04-15","arxiv_id":"2104.07282","repositories_listed":0,"syntology":null},{"url":null,"slug":"gan-based-interactive-reinforcement-learning","title":"GAN-Based Interactive Reinforcement Learning from Demonstration and Human Evaluative Feedback","date":"2021-04-14","arxiv_id":"2104.06600","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-comfort-aware-reinforcement-learning","title":"Visual Comfort Aware-Reinforcement Learning for Depth Adjustment of Stereoscopic 3D Images","date":"2021-04-14","arxiv_id":"2104.06782","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-reinforcement-learning-for","title":"Data-Driven Reinforcement Learning for Virtual Character Animation Control","date":"2021-04-13","arxiv_id":"2104.06358","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-the-long-term-average-reward-for","title":"Optimizing the Long-Term Average Reward for Continuing MDPs: A Technical Report","date":"2021-04-13","arxiv_id":"2104.06139","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-admission-control","title":"Reinforcement learning for Admission Control in 5G Wireless Networks","date":"2021-04-13","arxiv_id":"2104.10761","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-shaping-with-dynamic-trajectory","title":"Reward Shaping with Dynamic Trajectory Aggregation","date":"2021-04-13","arxiv_id":"2104.06163","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-shaping-with-subgoals-for-social","title":"Reward Shaping with Subgoals for Social Navigation","date":"2021-04-13","arxiv_id":"2104.06410","repositories_listed":0,"syntology":null},{"url":null,"slug":"subgoal-based-reward-shaping-to-improve","title":"Subgoal-based Reward Shaping to Improve Efficiency in Reinforcement Learning","date":"2021-04-13","arxiv_id":"2104.06411","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-stage-training-algorithm-for-ai-robot","title":"Two-stage training algorithm for AI robot soccer","date":"2021-04-13","arxiv_id":"2104.05931","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-controller","title":"Deep Reinforcement Learning Based Controller for Active Heave Compensation","date":"2021-04-12","arxiv_id":"2104.05599","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-matching-markets-in-power-grid","title":"Dynamic Matching Markets in Power Grid: Concepts and Solution using Deep Reinforcement Learning","date":"2021-04-12","arxiv_id":"2104.05654","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-on-reinforcement-learning-for-language","title":"Survey on reinforcement learning for language processing","date":"2021-04-12","arxiv_id":"2104.05565","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-goal-conditioned-policy-with-intrinsic-1","title":"Learn Goal-Conditioned Policy with Intrinsic Motivation for Deep Reinforcement Learning","date":"2021-04-11","arxiv_id":"2104.05043","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-based-energy","title":"A Reinforcement-Learning-Based Energy-Efficient Framework for Multi-Task Video Analytics Pipeline","date":"2021-04-09","arxiv_id":"2104.04443","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-a-control","title":"Inverse Reinforcement Learning: A Control Lyapunov Approach","date":"2021-04-09","arxiv_id":"2104.04483","repositories_listed":0,"syntology":null},{"url":null,"slug":"jamming-resilient-path-planning-for-multiple","title":"Jamming-Resilient Path Planning for Multiple UAVs via Deep Reinforcement Learning","date":"2021-04-09","arxiv_id":"2104.04477","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-sampling-policy-for-faster","title":"Learning Sampling Policy for Faster Derivative Free Optimization","date":"2021-04-09","arxiv_id":"2104.04405","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-reweight-imaginary-transitions","title":"Learning to Reweight Imaginary Transitions for Model-Based Reinforcement Learning","date":"2021-04-09","arxiv_id":"2104.04174","repositories_listed":0,"syntology":null},{"url":null,"slug":"symmetry-reduction-for-deep-reinforcement","title":"Symmetry reduction for deep reinforcement learning active control of chaotic spatiotemporal dynamics","date":"2021-04-09","arxiv_id":"2104.05437","repositories_listed":0,"syntology":null},{"url":null,"slug":"acerac-efficient-reinforcement-learning-in","title":"ACERAC: Efficient reinforcement learning in fine time discretization","date":"2021-04-08","arxiv_id":"2104.04004","repositories_listed":0,"syntology":null},{"url":null,"slug":"finite-sample-analysis-for-two-time-scale-non","title":"Non-Asymptotic Analysis for Two Time-scale TDC with General Smooth Function Approximation","date":"2021-04-07","arxiv_id":"2104.02836","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-a-disentangled","title":"Reinforcement Learning with a Disentangled Universal Value Function for Item Recommendation","date":"2021-04-07","arxiv_id":"2104.02981","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-visual-attention-and-invariance","title":"Unsupervised Visual Attention and Invariance for Reinforcement Learning","date":"2021-04-07","arxiv_id":"2104.02921","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-robust-nmpc-using-reinforcement","title":"Approximate Robust NMPC using Reinforcement Learning","date":"2021-04-06","arxiv_id":"2104.02743","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-simulation-of-ride-hailing","title":"Data-Driven Simulation of Ride-Hailing Services using Imitation and Reinforcement Learning","date":"2021-04-06","arxiv_id":"2104.02661","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-deep-reinforcement-learning-for-3","title":"Distributed Deep Reinforcement Learning for Collaborative Spectrum Sharing","date":"2021-04-06","arxiv_id":"2104.02059","repositories_listed":0,"syntology":null},{"url":null,"slug":"mpc-based-reinforcement-learning-for-economic","title":"MPC-based Reinforcement Learning for Economic Problems with Application to Battery Storage","date":"2021-04-06","arxiv_id":"2104.02411","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-extension-of-reinforcement","title":"Progressive extension of reinforcement learning action dimension for asymmetric assembly tasks","date":"2021-04-06","arxiv_id":"2104.04078","repositories_listed":0,"syntology":null},{"url":null,"slug":"zeus-efficiently-localizing-actions-in-videos","title":"Zeus: Efficiently Localizing Actions in Videos using Reinforcement Learning","date":"2021-04-06","arxiv_id":"2104.06142","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dual-critic-reinforcement-learning","title":"A Dual-Critic Reinforcement Learning Framework for Frame-level Bit Allocation in HEVC/H.265","date":"2021-04-05","arxiv_id":"2104.01735","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-monotonic-value-function-factorization","title":"NQMIX: Non-monotonic Value Function Factorization for Deep Multi-Agent Reinforcement Learning","date":"2021-04-05","arxiv_id":"2104.01939","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-transformers-in-reinforcement-1","title":"Efficient Transformers in Reinforcement Learning using Actor-Learner Distillation","date":"2021-04-04","arxiv_id":"2104.01655","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-minimizing-age-of","title":"Distributed Reinforcement Learning for Age of Information Minimization in Real-Time IoT Systems","date":"2021-04-04","arxiv_id":"2104.01527","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dynamics-perspective-of-pursuit-evasion","title":"A Dynamics Perspective of Pursuit-Evasion Games of Intelligent Agents with the Ability to Learn","date":"2021-04-03","arxiv_id":"2104.01445","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-powered-irs","title":"Deep Reinforcement Learning Powered IRS-Assisted Downlink NOMA","date":"2021-04-03","arxiv_id":"2104.01414","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-emotional-text-to","title":"Reinforcement Learning for Emotional Text-to-Speech Synthesis with Improved Emotion Discriminability","date":"2021-04-03","arxiv_id":"2104.01408","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-double-deep-q-learning-for-joint","title":"Federated Double Deep Q-learning for Joint Delay and Energy Minimization in IoT networks","date":"2021-04-02","arxiv_id":"2104.11320","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-dose-helical-cbct-denoising-by-using","title":"Low Dose Helical CBCT denoising by using domain filtering with deep reinforcement learning","date":"2021-04-02","arxiv_id":"2104.00889","repositories_listed":0,"syntology":null},{"url":null,"slug":"dealio-data-efficient-adversarial-learning","title":"DEALIO: Data-Efficient Adversarial Learning for Imitation from Observation","date":"2021-03-31","arxiv_id":"2104.00163","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-constrained","title":"Deep Reinforcement Learning for Constrained Field Development Optimization in Subsurface Two-phase Flow","date":"2021-03-31","arxiv_id":"2104.00527","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-efficient-edge-computing-when-lyapunov","title":"Energy Efficient Edge Computing: When Lyapunov Meets Distributed Reinforcement Learning","date":"2021-03-31","arxiv_id":"2103.16985","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-reinforcement-learning-for","title":"Generalized Reinforcement Learning for Building Control using Behavioral Cloning","date":"2021-03-31","arxiv_id":"2104.00123","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlad-time-series-anomaly-detection-through","title":"RLAD: Time Series Anomaly Detection through Reinforcement Learning and Active Learning","date":"2021-03-31","arxiv_id":"2104.00543","repositories_listed":0,"syntology":null},{"url":null,"slug":"solving-heterogeneous-general-equilibrium","title":"Solving Heterogeneous General Equilibrium Economic Models with Deep Reinforcement Learning","date":"2021-03-31","arxiv_id":"2103.16977","repositories_listed":0,"syntology":null},{"url":null,"slug":"fair-iot-fairness-aware-human-in-the-loop","title":"FaiR-IoT: Fairness-aware Human-in-the-Loop Reinforcement Learning for Harnessing Human Variability in Personalized IoT","date":"2021-03-30","arxiv_id":"2103.16033","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-policies-for-real-time-control-using","title":"Online Policies for Real-Time Control Using MRAC-RL","date":"2021-03-30","arxiv_id":"2103.16551","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-optimization-of-1","title":"Reinforcement learning for optimization of variational quantum circuit architectures","date":"2021-03-30","arxiv_id":"2103.16089","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-automated-game-testing-with-deep","title":"Augmenting Automated Game Testing with Deep Reinforcement Learning","date":"2021-03-29","arxiv_id":"2103.15819","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-hedging-of-derivatives-using","title":"Deep Hedging of Derivatives Using Reinforcement Learning","date":"2021-03-29","arxiv_id":"2103.16409","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-of-event","title":"Deep reinforcement learning of event-triggered communication and control for multi-agent cooperative transport","date":"2021-03-29","arxiv_id":"2103.15260","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-resource-management-for-mc-noma-a-deep","title":"Joint Resource Management for MC-NOMA: A Deep Reinforcement Learning Approach","date":"2021-03-29","arxiv_id":"2103.15371","repositories_listed":0,"syntology":null},{"url":null,"slug":"laser-learning-a-latent-action-space-for","title":"LASER: Learning a Latent Action Space for Efficient Reinforcement Learning","date":"2021-03-29","arxiv_id":"2103.15793","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-sample-efficiency-and","title":"Measuring Sample Efficiency and Generalization in Reinforcement Learning Benchmarks: NeurIPS 2020 Procgen Benchmark","date":"2021-03-29","arxiv_id":"2103.15332","repositories_listed":0,"syntology":null},{"url":null,"slug":"ph-rl-a-personalization-architecture-to","title":"pH-RL: A personalization architecture to bring reinforcement learning to health practice","date":"2021-03-29","arxiv_id":"2103.15908","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-beyond-expectation","title":"Reinforcement Learning Beyond Expectation","date":"2021-03-29","arxiv_id":"2104.00540","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowru-knowledge-reusing-via-knowledge","title":"KnowRU: Knowledge Reusing via Knowledge Distillation in Multi-agent Reinforcement Learning","date":"2021-03-27","arxiv_id":"2103.14891","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-convex-programming-approach-to-data-driven","title":"A Convex Programming Approach to Data-Driven Risk-Averse Reinforcement Learning","date":"2021-03-26","arxiv_id":"2103.14606","repositories_listed":0,"syntology":null}],"record_sha256":"7cff52a1dec9745b3ddd42ad080c095f8efa665a2aa5bec733a8a6f1ecaf923e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}