{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/93","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":93,"pages_in_order":135,"rows_per_page":100,"rows":[9201,9300],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/92","next":"/task/reinforcement-learning-2/papers/94","papers":[{"url":null,"slug":"model-selection-with-near-optimal-rates-for","title":"Model Selection for Generic Reinforcement Learning","date":"2021-07-13","arxiv_id":"2107.05849","repositories_listed":0,"syntology":null},{"url":null,"slug":"pessimistic-model-based-offline-rl-pac-bounds","title":"Pessimistic Model-based Offline Reinforcement Learning under Partial Coverage","date":"2021-07-13","arxiv_id":"2107.06226","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simple-reward-free-approach-to-constrained","title":"A Simple Reward-free Approach to Constrained Reinforcement Learning","date":"2021-07-12","arxiv_id":"2107.05216","repositories_listed":0,"syntology":null},{"url":null,"slug":"polynomial-time-reinforcement-learning-in","title":"Polynomial Time Reinforcement Learning in Factored State MDPs with Linear Value Functions","date":"2021-07-12","arxiv_id":"2107.05187","repositories_listed":0,"syntology":null},{"url":"/paper/r3l-connecting-deep-reinforcement-learning-to","slug":"r3l-connecting-deep-reinforcement-learning-to","title":"R3L: Connecting Deep Reinforcement Learning to Recurrent Neural Networks for Image Denoising via Residual Recovery","date":"2021-07-12","arxiv_id":"2107.05318","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-proactive","title":"Reinforcement Learning based Proactive Control for Transmission Grid Resilience to Wildfire","date":"2021-07-12","arxiv_id":"2107.05756","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-stable-molecules-using-imitation","title":"Generating stable molecules using imitation and reinforcement learning","date":"2021-07-11","arxiv_id":"2107.05007","repositories_listed":0,"syntology":null},{"url":null,"slug":"attend2pack-bin-packing-through-deep","title":"Attend2Pack: Bin Packing through Deep Reinforcement Learning with Attention","date":"2021-07-09","arxiv_id":"2107.04333","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-probabilistic-reward-machines-from","title":"Inferring Probabilistic Reward Machines from Non-Markovian Reward Processes for Reinforcement Learning","date":"2021-07-09","arxiv_id":"2107.04633","repositories_listed":0,"syntology":null},{"url":null,"slug":"likelihood-ratio-based-policy-gradient","title":"Policy Gradient Methods for Distortion Risk Measures","date":"2021-07-09","arxiv_id":"2107.04422","repositories_listed":0,"syntology":null},{"url":null,"slug":"nvcell-standard-cell-layout-in-advanced","title":"NVCell: Standard Cell Layout in Advanced Technology Nodes with Reinforcement Learning","date":"2021-07-09","arxiv_id":"2107.07044","repositories_listed":0,"syntology":null},{"url":null,"slug":"offline-reinforcement-learning-with-1","title":"Offline reinforcement learning with uncertainty for treatment strategies in sepsis","date":"2021-07-09","arxiv_id":"2107.04491","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-hybrid-genetic-algorithm-for-the","title":"Reinforced Hybrid Genetic Algorithm for the Traveling Salesman Problem","date":"2021-07-09","arxiv_id":"2107.06870","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptation-of-quadruped-robot-locomotion-with","title":"Adaptation of Quadruped Robot Locomotion with Meta-Learning","date":"2021-07-08","arxiv_id":"2107.03741","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-stress-testing-for-adversarial","title":"Adaptive Stress Testing for Adversarial Learning in a Financial Environment","date":"2021-07-08","arxiv_id":"2107.03577","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-gain-control-through-deep","title":"Automated Gain Control Through Deep Reinforcement Learning for Downstream Radar Object Detection","date":"2021-07-08","arxiv_id":"2107.03792","repositories_listed":0,"syntology":null},{"url":null,"slug":"claim-curriculum-learning-policy-for","title":"CLAIM: Curriculum Learning Policy for Influence Maximization in Unknown Social Networks","date":"2021-07-08","arxiv_id":"2107.03603","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-model-based-multi-agent-mean-field","title":"Efficient Model-Based Multi-Agent Mean-Field Reinforcement Learning","date":"2021-07-08","arxiv_id":"2107.04050","repositories_listed":0,"syntology":null},{"url":null,"slug":"sublinear-regret-for-learning-pomdps","title":"Sublinear Regret for Learning POMDPs","date":"2021-07-08","arxiv_id":"2107.03635","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-autonomous-pipeline-inspection-with","title":"Towards Autonomous Pipeline Inspection with Hierarchical Reinforcement Learning","date":"2021-07-08","arxiv_id":"2107.03685","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-model-search-via-reinforcement","title":"Federated Model Search via Reinforcement Learning","date":"2021-07-07","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-time-invariant-reward-functions","title":"Learning Time-Invariant Reward Functions through Model-Based Inverse Reinforcement Learning","date":"2021-07-07","arxiv_id":"2107.03186","repositories_listed":0,"syntology":null},{"url":null,"slug":"pseudo-model-free-hedging-for-variable","title":"Pseudo-Model-Free Hedging for Variable Annuities via Deep Reinforcement Learning","date":"2021-07-07","arxiv_id":"2107.03340","repositories_listed":0,"syntology":null},{"url":null,"slug":"quadruped-locomotion-on-non-rigid-terrain","title":"Quadruped Locomotion on Non-Rigid Terrain using Reinforcement Learning","date":"2021-07-07","arxiv_id":"2107.02955","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-short-note-on-the-relationship-of","title":"A Short Note on the Relationship of Information Gain and Eluder Dimension","date":"2021-07-06","arxiv_id":"2107.02377","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-for-heuristic","title":"Meta-Reinforcement Learning for Heuristic Planning","date":"2021-07-06","arxiv_id":"2107.02603","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-explainable-artificial","title":"A Review of Explainable Artificial Intelligence in Manufacturing","date":"2021-07-05","arxiv_id":"2107.02295","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-of-rough-terrain-vehicles-using-deep","title":"Control of rough terrain vehicles using deep reinforcement learning","date":"2021-07-05","arxiv_id":"2107.01867","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-least-restriction-for-offline","title":"The Least Restriction for Offline Reinforcement Learning","date":"2021-07-05","arxiv_id":"2107.01757","repositories_listed":0,"syntology":null},{"url":null,"slug":"winning-at-any-cost-infringing-the-cartel","title":"Winning at Any Cost -- Infringing the Cartel Prohibition With Reinforcement Learning","date":"2021-07-05","arxiv_id":"2107.01856","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-dimensional-state-and-action","title":"Low-Dimensional State and Action Representation Learning with MDP Homomorphism Metrics","date":"2021-07-04","arxiv_id":"2107.01677","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-dimensional-state-representation-learning-1","title":"Low Dimensional State Representation Learning with Robotics Priors in Continuous Action Spaces","date":"2021-07-04","arxiv_id":"2107.01667","repositories_listed":0,"syntology":null},{"url":null,"slug":"examining-average-and-discounted-reward","title":"Examining average and discounted reward optimality criteria in reinforcement learning","date":"2021-07-03","arxiv_id":"2107.01348","repositories_listed":0,"syntology":null},{"url":null,"slug":"traffic-signal-control-with-communicative","title":"Traffic Signal Control with Communicative Deep Reinforcement Learning Agents: a Case Study","date":"2021-07-03","arxiv_id":"2107.01347","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-deep-reinforcement-learning-based","title":"A Novel Deep Reinforcement Learning Based Stock Direction Prediction using Knowledge Graph and Community Aware Sentiments","date":"2021-07-02","arxiv_id":"2107.00931","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-value-function-gaps-improved-instance","title":"Beyond Value-Function Gaps: Improved Instance-Dependent Regret Bounds for Episodic Reinforcement Learning","date":"2021-07-02","arxiv_id":"2107.01264","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-feedback-enabled","title":"Reinforcement Learning for Feedback-Enabled Cyber Resilience","date":"2021-07-02","arxiv_id":"2107.00783","repositories_listed":0,"syntology":null},{"url":null,"slug":"socialai-benchmarking-socio-cognitive","title":"SocialAI: Benchmarking Socio-Cognitive Abilities in Deep Reinforcement Learning Agents","date":"2021-07-02","arxiv_id":"2107.00956","repositories_listed":0,"syntology":null},{"url":null,"slug":"blending-task-success-and-user-satisfaction","title":"Blending Task Success and User Satisfaction: Analysis of Learned Dialogue Behaviour with Multiple Rewards","date":"2021-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-conditioned-reinforcement-learning-with","title":"Goal-Conditioned Reinforcement Learning with Imagined Subgoals","date":"2021-07-01","arxiv_id":"2107.00541","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-based","title":"Inverse Reinforcement Learning Based Stochastic Driver Behavior Learning","date":"2021-07-01","arxiv_id":"2107.06344","repositories_listed":0,"syntology":null},{"url":null,"slug":"mher-model-based-hindsight-experience-replay","title":"MHER: Model-based Hindsight Experience Replay","date":"2021-07-01","arxiv_id":"2107.00306","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-power-allocation-for-rate-splitting","title":"Optimal Power Allocation for Rate Splitting Communications with Deep Reinforcement Learning","date":"2021-07-01","arxiv_id":"2107.00238","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-stochastic-admm-for-decentralized","title":"Adaptive Stochastic ADMM for Decentralized Reinforcement Learning in Edge Industrial IoT","date":"2021-06-30","arxiv_id":"2107.00481","repositories_listed":0,"syntology":null},{"url":null,"slug":"decomposing-the-prediction-problem-autonomous","title":"Decomposing the Prediction Problem; Autonomous Navigation by neoRL Agents","date":"2021-06-30","arxiv_id":"2106.15868","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-design-of-grating-couplers-using-the","title":"Inverse Design of Grating Couplers Using the Policy Gradient Method from Reinforcement Learning","date":"2021-06-30","arxiv_id":"2107.00088","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-disease","title":"Reinforcement Learning based Disease Progression Model for Alzheimer's Disease","date":"2021-06-30","arxiv_id":"2106.16187","repositories_listed":0,"syntology":null},{"url":null,"slug":"drill-deep-reinforcement-learning-for","title":"DRILL-- Deep Reinforcement Learning for Refinement Operators in $\\mathcal{ALC}$","date":"2021-06-29","arxiv_id":"2106.15373","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalization-of-reinforcement-learning-with","title":"Generalization of Reinforcement Learning with Policy-Aware Adversarial Data Augmentation","date":"2021-06-29","arxiv_id":"2106.15587","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiagent-deep-reinforcement-learning","title":"Deep Multiagent Reinforcement Learning: Challenges and Directions","date":"2021-06-29","arxiv_id":"2106.15691","repositories_listed":0,"syntology":null},{"url":null,"slug":"structure-aware-reinforcement-learning-for","title":"Structure-aware reinforcement learning for node-overload protection in mobile edge computing","date":"2021-06-29","arxiv_id":"2107.01025","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-model-predictive-and","title":"Data-driven Model Predictive and Reinforcement Learning Based Control for Building Energy Management: a Survey","date":"2021-06-28","arxiv_id":"2106.14450","repositories_listed":0,"syntology":null},{"url":null,"slug":"expert-q-learning-deep-q-learning-with-state","title":"Expert Q-learning: Deep Reinforcement Learning with Coarse State Values from Offline Expert Examples","date":"2021-06-28","arxiv_id":"2106.14642","repositories_listed":0,"syntology":null},{"url":null,"slug":"modularity-in-reinforcement-learning-via","title":"Modularity in Reinforcement Learning via Algorithmic Independence in Credit Assignment","date":"2021-06-28","arxiv_id":"2106.14993","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-for-4","title":"A Reinforcement Learning Approach for Sequential Spatial Transformer Networks","date":"2021-06-27","arxiv_id":"2106.14295","repositories_listed":0,"syntology":null},{"url":null,"slug":"concentration-of-contractive-stochastic","title":"Concentration of Contractive Stochastic Approximation and Reinforcement Learning","date":"2021-06-27","arxiv_id":"2106.14308","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-control-with-deep-reinforcement-1","title":"Continuous Control with Deep Reinforcement Learning for Autonomous Vessels","date":"2021-06-27","arxiv_id":"2106.14130","repositories_listed":0,"syntology":null},{"url":null,"slug":"regret-analysis-in-deterministic","title":"Regret Analysis in Deterministic Reinforcement Learning","date":"2021-06-27","arxiv_id":"2106.14338","repositories_listed":0,"syntology":null},{"url":null,"slug":"discovering-generalizable-skills-via","title":"Discovering Generalizable Skills via Automated Generation of Diverse Tasks","date":"2021-06-26","arxiv_id":"2106.13935","repositories_listed":0,"syntology":null},{"url":null,"slug":"intrinsically-motivated-self-supervised","title":"Intrinsically Motivated Self-supervised Learning in Reinforcement Learning","date":"2021-06-26","arxiv_id":"2106.13970","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-accuracy-and-fairness-for","title":"Balancing Accuracy and Fairness for Interactive Recommendation with Reinforcement Learning","date":"2021-06-25","arxiv_id":"2106.13386","repositories_listed":0,"syntology":null},{"url":null,"slug":"branch-prediction-as-a-reinforcement-learning","title":"Branch Prediction as a Reinforcement Learning Problem: Why, How and Case Studies","date":"2021-06-25","arxiv_id":"2106.13429","repositories_listed":0,"syntology":null},{"url":null,"slug":"predictive-control-using-learned-state-space","title":"Predictive Control Using Learned State Space Models via Rolling Horizon Evolution","date":"2021-06-25","arxiv_id":"2106.13911","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-mean-field-games","title":"Reinforcement Learning for Mean Field Games, with Applications to Economics","date":"2021-06-25","arxiv_id":"2106.13755","repositories_listed":0,"syntology":null},{"url":null,"slug":"density-constrained-reinforcement-learning-1","title":"Density Constrained Reinforcement Learning","date":"2021-06-24","arxiv_id":"2106.12764","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-robot-deep-reinforcement-learning-for","title":"Hierarchically Integrated Models: Learning to Navigate from Heterogeneous Robots","date":"2021-06-24","arxiv_id":"2106.13280","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-option-keyboard-combining-skills-in-1","title":"The Option Keyboard: Combining Skills in Reinforcement Learning","date":"2021-06-24","arxiv_id":"2106.13105","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolving-hierarchical-memory-prediction","title":"Evolving Hierarchical Memory-Prediction Machines in Multi-Task Reinforcement Learning","date":"2021-06-23","arxiv_id":"2106.12659","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-model-based-reinforcement","title":"Uncertainty-Aware Model-Based Reinforcement Learning with Application to Autonomous Driving","date":"2021-06-23","arxiv_id":"2106.12194","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-framework-for-conservative","title":"A Reduction-Based Framework for Conservative Bandits and Reinforcement Learning","date":"2021-06-22","arxiv_id":"2106.11692","repositories_listed":0,"syntology":null},{"url":null,"slug":"agnostic-reinforcement-learning-with-low-rank","title":"Agnostic Reinforcement Learning with Low-Rank MDPs and Rich Observations","date":"2021-06-22","arxiv_id":"2106.11519","repositories_listed":0,"syntology":null},{"url":null,"slug":"lifted-model-checking-for-relational-mdps","title":"Lifted Model Checking for Relational MDPs","date":"2021-06-22","arxiv_id":"2106.11735","repositories_listed":0,"syntology":null},{"url":null,"slug":"mmd-mix-value-function-factorisation-with","title":"MMD-MIX: Value Function Factorisation with Maximum Mean Discrepancy for Cooperative Multi-Agent Reinforcement Learning","date":"2021-06-22","arxiv_id":"2106.11652","repositories_listed":0,"syntology":null},{"url":null,"slug":"off-policy-reinforcement-learning-with","title":"Off-Policy Reinforcement Learning with Delayed Rewards","date":"2021-06-22","arxiv_id":"2106.11854","repositories_listed":0,"syntology":null},{"url":null,"slug":"provably-efficient-representation-learning-in","title":"Provably Efficient Representation Selection in Low-rank Markov Decision Processes: From Online to Offline RL","date":"2021-06-22","arxiv_id":"2106.11935","repositories_listed":0,"syntology":null},{"url":null,"slug":"uniform-pac-bounds-for-reinforcement-learning","title":"Uniform-PAC Bounds for Reinforcement Learning with Linear Function Approximation","date":"2021-06-22","arxiv_id":"2106.11612","repositories_listed":0,"syntology":null},{"url":null,"slug":"analytically-tractable-bayesian-deep-q","title":"Analytically Tractable Bayesian Deep Q-Learning","date":"2021-06-21","arxiv_id":"2106.11086","repositories_listed":0,"syntology":null},{"url":null,"slug":"cogment-open-source-framework-for-distributed","title":"Cogment: Open Source Framework For Distributed Multi-actor Training, Deployment & Operations","date":"2021-06-21","arxiv_id":"2106.11345","repositories_listed":0,"syntology":null},{"url":null,"slug":"emphatic-algorithms-for-deep-reinforcement","title":"Emphatic Algorithms for Deep Reinforcement Learning","date":"2021-06-21","arxiv_id":"2106.11779","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-model-based-hierarchical","title":"Interpretable Model-based Hierarchical Reinforcement Learning using Inductive Logic Programming","date":"2021-06-21","arxiv_id":"2106.11417","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-smoothing-for-provably-robust","title":"Policy Smoothing for Provably Robust Reinforcement Learning","date":"2021-06-21","arxiv_id":"2106.11420","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-resource-1","title":"Reinforcement Learning for Resource Allocation in Steerable Laser-based Optical Wireless Systems","date":"2021-06-21","arxiv_id":"2106.11368","repositories_listed":0,"syntology":null},{"url":null,"slug":"scientific-multi-agent-reinforcement-learning","title":"Scientific multi-agent reinforcement learning for wall-models of turbulent flows","date":"2021-06-21","arxiv_id":"2106.11144","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-offline-reinforcement-learning-with","title":"Boosting Offline Reinforcement Learning with Residual Generative Modeling","date":"2021-06-19","arxiv_id":"2106.10411","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-learning-for-robust-fitting-a-1","title":"Unsupervised Learning for Robust Fitting: A Reinforcement Learning Approach","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"video-summarization-through-reinforcement","title":"Video Summarization through Reinforcement Learning with a 3D Spatio-Temporal U-Net","date":"2021-06-19","arxiv_id":"2106.10528","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarially-trained-neural-policies-in-the","title":"Adversarially Trained Neural Policies in the Fourier Domain","date":"2021-06-18","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-models-predict","title":"Deep Reinforcement Learning Models Predict Visual Responses in the Brain: A Preliminary Result","date":"2021-06-18","arxiv_id":"2106.10112","repositories_listed":0,"syntology":null},{"url":null,"slug":"goal-directed-planning-by-reinforcement","title":"Goal-Directed Planning by Reinforcement Learning and Active Inference","date":"2021-06-18","arxiv_id":"2106.09938","repositories_listed":0,"syntology":null},{"url":null,"slug":"non-robust-feature-mapping-in-deep","title":"Non-Robust Feature Mapping in Deep Reinforcement Learning","date":"2021-06-18","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-sample-complexity-of-batch","title":"The Curse of Passive Data Collection in Batch Reinforcement Learning","date":"2021-06-18","arxiv_id":"2106.09973","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-efficient-social-navigation-using","title":"Sample Efficient Social Navigation Using Inverse Reinforcement Learning","date":"2021-06-18","arxiv_id":"2106.10318","repositories_listed":0,"syntology":null},{"url":null,"slug":"scenic4rl-programmatic-modeling-and","title":"Scenic4RL: Programmatic Modeling and Generation of Reinforcement Learning Environments","date":"2021-06-18","arxiv_id":"2106.10365","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategically-timed-state-observation-attacks","title":"Strategically-timed State-Observation Attacks on Deep Reinforcement Learning Agents","date":"2021-06-18","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach","title":"A Deep Reinforcement Learning Approach towards Pendulum Swing-up Problem based on TF-Agents","date":"2021-06-17","arxiv_id":"2106.09556","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-approach-for-an-irs","title":"A Reinforcement Learning Approach for an IRS-assisted NOMA Network","date":"2021-06-17","arxiv_id":"2106.09611","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-the-function-approximation","title":"Adapting the Function Approximation Architecture in Online Reinforcement Learning","date":"2021-06-17","arxiv_id":"2106.09776","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-1","title":"Deep Reinforcement Learning Based Optimization for IRS Based UAV-NOMA Downlink Networks","date":"2021-06-17","arxiv_id":"2106.09616","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-automated","title":"Deep reinforcement learning with automated label extraction from clinical reports accurately classifies 3D MRI brain volumes","date":"2021-06-17","arxiv_id":"2106.09812","repositories_listed":0,"syntology":null},{"url":null,"slug":"many-agent-reinforcement-learning-under","title":"Many Agent Reinforcement Learning Under Partial Observability","date":"2021-06-17","arxiv_id":"2106.09825","repositories_listed":0,"syntology":null}],"record_sha256":"f32425a46269a1fde2eff4ebe21d7633a90edd39dd9a28eade5e8752d1685b67","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}