{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/114","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":114,"pages_in_order":132,"rows_per_page":100,"rows":[11301,11400],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/113","next":"/task/reinforcement-learning/papers/115","papers":[{"url":null,"slug":"q-learning-with-ucb-exploration-is-sample","title":"Q-learning with UCB Exploration is Sample Efficient for Infinite-Horizon MDP","date":"2019-01-27","arxiv_id":"1901.09311","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-shaping-via-meta-learning","title":"Reward Shaping via Meta-Learning","date":"2019-01-27","arxiv_id":"1901.09330","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-propagation-for-decentralized-networked","title":"Value Propagation for Decentralized Networked Deep Multi-agent Reinforcement Learning","date":"2019-01-27","arxiv_id":"1901.09326","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-recursive-reasoning-for-multi","title":"Probabilistic Recursive Reasoning for Multi-Agent Reinforcement Learning","date":"2019-01-26","arxiv_id":"1901.09207","repositories_listed":0,"syntology":null},{"url":null,"slug":"almost-boltzmann-exploration","title":"Almost Boltzmann Exploration","date":"2019-01-25","arxiv_id":"1901.08708","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-policy-iteration-for-scalable","title":"Distributed Policy Iteration for Scalable Approximation of Cooperative Multi-Agent Policies","date":"2019-01-25","arxiv_id":"1901.08761","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-deep-reinforcement-learning-for","title":"Model-based Deep Reinforcement Learning for Dynamic Portfolio Optimization","date":"2019-01-25","arxiv_id":"1901.08740","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-reinforcement-learning","title":"Federated Deep Reinforcement Learning","date":"2019-01-24","arxiv_id":"1901.08277","repositories_listed":0,"syntology":null},{"url":null,"slug":"feudal-multi-agent-hierarchies-for","title":"Feudal Multi-Agent Hierarchies for Cooperative Reinforcement Learning","date":"2019-01-24","arxiv_id":"1901.08492","repositories_listed":0,"syntology":null},{"url":null,"slug":"never-forget-balancing-exploration-and","title":"Never Forget: Balancing Exploration and Exploitation via Learning Optical Flow","date":"2019-01-24","arxiv_id":"1901.08486","repositories_listed":0,"syntology":null},{"url":null,"slug":"sample-complexity-of-estimating-the-policy","title":"Sample Complexity of Estimating the Policy Gradient for Nearly Deterministic Dynamical Systems","date":"2019-01-24","arxiv_id":"1901.08562","repositories_listed":0,"syntology":null},{"url":null,"slug":"thirty-years-of-machine-learningthe-road-to","title":"Thirty Years of Machine Learning: The Road to Pareto-Optimal Wireless Networks","date":"2019-01-24","arxiv_id":"1902.01946","repositories_listed":0,"syntology":null},{"url":null,"slug":"distillation-strategies-for-proximal-policy","title":"Distillation Strategies for Proximal Policy Optimization","date":"2019-01-23","arxiv_id":"1901.08128","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-reinforcement-learning-for-multi","title":"Hierarchical Reinforcement Learning for Multi-agent MOBA Game","date":"2019-01-23","arxiv_id":"1901.08004","repositories_listed":0,"syntology":null},{"url":null,"slug":"phonetic-enriched-text-representation-for","title":"Phonetic-enriched Text Representation for Chinese Sentiment Analysis with Reinforcement Learning","date":"2019-01-23","arxiv_id":"1901.07880","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-of-markov-decision","title":"Reinforcement Learning of Markov Decision Processes with Peak Constraints","date":"2019-01-23","arxiv_id":"1901.07839","repositories_listed":0,"syntology":null},{"url":null,"slug":"trust-region-value-optimization-using-kalman","title":"Trust Region Value Optimization using Kalman Filtering","date":"2019-01-23","arxiv_id":"1901.07860","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-recovery-controller-for-a-quadrupedal","title":"Robust Recovery Controller for a Quadrupedal Robot using Deep Reinforcement Learning","date":"2019-01-22","arxiv_id":"1901.07517","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-imitation-learning-with-recurrent","title":"Towards Learning to Imitate from a Single Video Demonstration","date":"2019-01-22","arxiv_id":"1901.07186","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-short-survey-on-probabilistic-reinforcement","title":"A Short Survey on Probabilistic Reinforcement Learning","date":"2019-01-21","arxiv_id":"1901.07010","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-retrosynthetic-planning-through-self","title":"Learning retrosynthetic planning through self-play","date":"2019-01-19","arxiv_id":"1901.06569","repositories_listed":0,"syntology":null},{"url":null,"slug":"lifelong-federated-reinforcement-learning-a","title":"Lifelong Federated Reinforcement Learning: A Learning Architecture for Navigation in Cloud Robotic Systems","date":"2019-01-19","arxiv_id":"1901.06455","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-physically-safe-reinforcement","title":"Towards Physically Safe Reinforcement Learning under Supervision","date":"2019-01-19","arxiv_id":"1901.06576","repositories_listed":0,"syntology":null},{"url":null,"slug":"theory-of-minds-understanding-behavior-in","title":"Theory of Minds: Understanding Behavior in Groups Through Inverse Planning","date":"2019-01-18","arxiv_id":"1901.06085","repositories_listed":0,"syntology":null},{"url":null,"slug":"codex-bit-flexible-encoding-for-streaming","title":"CodeX: Bit-Flexible Encoding for Streaming-based FPGA Acceleration of DNNs","date":"2019-01-17","arxiv_id":"1901.05582","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-embedded","title":"Multi-agent Reinforcement Learning Embedded Game for the Optimization of Building Energy Control and Power System Planning","date":"2019-01-17","arxiv_id":"1901.07333","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionarily-curated-curriculum-learning","title":"Evolutionarily-Curated Curriculum Learning for Deep Reinforcement Learning Agents","date":"2019-01-16","arxiv_id":"1901.05431","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-learning-on-graphs-a","title":"Representation Learning on Graphs: A Reinforcement Learning Application","date":"2019-01-16","arxiv_id":"1901.05351","repositories_listed":0,"syntology":null},{"url":null,"slug":"comparing-knowledge-based-reinforcement","title":"Comparing Knowledge-based Reinforcement Learning to Neural Networks in a Strategy Game","date":"2019-01-15","arxiv_id":"1901.04626","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-sepsis-treatment-strategies-by","title":"Improving Sepsis Treatment Strategies by Combining Deep and Kernel-Based Reinforcement Learning","date":"2019-01-15","arxiv_id":"1901.04670","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-level-control-of-a-quadrotor-with-deep","title":"Low Level Control of a Quadrotor with Deep Model-Based Reinforcement Learning","date":"2019-01-11","arxiv_id":"1901.03737","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-power-neuromorphic-hardware-for-signal","title":"Low-Power Neuromorphic Hardware for Signal Processing Applications","date":"2019-01-11","arxiv_id":"1901.03690","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-global-convergence-of-imitation","title":"On the Global Convergence of Imitation Learning: A Case for Linear Quadratic Regulator","date":"2019-01-11","arxiv_id":"1901.03674","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-tensioning-method-using-deep","title":"A New Tensioning Method using Deep Reinforcement Learning for Surgical Pattern Cutting","date":"2019-01-10","arxiv_id":"1901.03327","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-perception-in-reinforcement-learning","title":"Motion Perception in Reinforcement Learning with Dynamic Objects","date":"2019-01-10","arxiv_id":"1901.03162","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-based-out-of-distribution","title":"Uncertainty-Based Out-of-Distribution Detection in Deep Reinforcement Learning","date":"2019-01-08","arxiv_id":"1901.02219","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dual-mode-adaptive-basal-bolus-advisor","title":"A dual mode adaptive basal-bolus advisor based on reinforcement learning","date":"2019-01-07","arxiv_id":"1901.01816","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-tree-search-for-portfolio-management","title":"A* Tree Search for Portfolio Management","date":"2019-01-07","arxiv_id":"1901.01855","repositories_listed":0,"syntology":null},{"url":null,"slug":"credit-assignment-techniques-in-stochastic","title":"Credit Assignment Techniques in Stochastic Computation Graphs","date":"2019-01-07","arxiv_id":"1901.01761","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-decentralized-autonomous-multiagent","title":"Towards a Decentralized, Autonomous Multiagent Framework for Mitigating Crop Loss","date":"2019-01-07","arxiv_id":"1901.02035","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-applications-of-deep-reinforcement","title":"Exploring applications of deep reinforcement learning for real-world autonomous driving systems","date":"2019-01-06","arxiv_id":"1901.01536","repositories_listed":0,"syntology":null},{"url":null,"slug":"recurrent-control-nets-for-deep-reinforcement","title":"Recurrent Control Nets for Deep Reinforcement Learning","date":"2019-01-06","arxiv_id":"1901.01994","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-should-i-do-now-marrying-reinforcement","title":"What Should I Do Now? Marrying Reinforcement Learning and Symbolic Planning","date":"2019-01-06","arxiv_id":"1901.01492","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-goal-directed-reinforcement","title":"Accelerating Goal-Directed Reinforcement Learning by Model Characterization","date":"2019-01-04","arxiv_id":"1901.01977","repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-teaching-in-hierarchical-genetic","title":"Machine Teaching in Hierarchical Genetic Reinforcement Learning: Curriculum Design of Reward Functions for Swarm Shepherding","date":"2019-01-04","arxiv_id":"1901.00949","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-decision-making-in-mixed-agent","title":"Optimal Decision-Making in Mixed-Agent Partially Observable Stochastic Environments via Reinforcement Learning","date":"2019-01-04","arxiv_id":"1901.01325","repositories_listed":0,"syntology":null},{"url":null,"slug":"qflow-a-reinforcement-learning-approach-to","title":"QFlow: A Learning Approach to High QoE Video Streaming at the Wireless Edge","date":"2019-01-04","arxiv_id":"1901.00959","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-computational-framework-for-motor-skill","title":"A Computational Framework for Motor Skill Acquisition","date":"2019-01-03","arxiv_id":"1901.01856","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-like-autonomous-car-following-model","title":"Human-Like Autonomous Car-Following Model with Deep Reinforcement Learning","date":"2019-01-03","arxiv_id":"1901.00569","repositories_listed":0,"syntology":null},{"url":null,"slug":"imminent-collision-mitigation-with","title":"Imminent Collision Mitigation with Reinforcement Learning and Vision","date":"2019-01-03","arxiv_id":"1901.00898","repositories_listed":0,"syntology":null},{"url":null,"slug":"natively-interpretable-machine-learning-and","title":"Natively Interpretable Machine Learning and Artificial Intelligence: Preliminary Results and Future Directions","date":"2019-01-02","arxiv_id":"1901.00246","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-theoretical-analysis-of-deep-q-learning","title":"A Theoretical Analysis of Deep Q-Learning","date":"2019-01-01","arxiv_id":"1901.00137","repositories_listed":0,"syntology":null},{"url":null,"slug":"complementary-reinforcement-learning-towards","title":"Complementary reinforcement learning towards explainable agents","date":"2019-01-01","arxiv_id":"1901.00188","repositories_listed":0,"syntology":null},{"url":null,"slug":"tighter-problem-dependent-regret-bounds-in","title":"Tighter Problem-Dependent Regret Bounds in Reinforcement Learning without Domain Knowledge using Value Function Bounds","date":"2019-01-01","arxiv_id":"1901.00210","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-multi-agent","title":"Deep Reinforcement Learning for Multi-Agent Systems: A Review of Challenges, Solutions and Applications","date":"2018-12-31","arxiv_id":"1812.11794","repositories_listed":0,"syntology":null},{"url":null,"slug":"stealing-neural-networks-via-timing-side","title":"Stealing Neural Networks via Timing Side Channels","date":"2018-12-31","arxiv_id":"1812.11720","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-reinforcement-learning-with-distribution","title":"Meta Reinforcement Learning with Distribution of Exploration Parameters Learned by Evolution Strategies","date":"2018-12-29","arxiv_id":"1812.11314","repositories_listed":0,"syntology":null},{"url":null,"slug":"differential-temporal-difference-learning","title":"Differential Temporal Difference Learning","date":"2018-12-28","arxiv_id":"1812.11137","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-planning-networks","title":"Dynamic Planning Networks","date":"2018-12-28","arxiv_id":"1812.11240","repositories_listed":0,"syntology":null},{"url":null,"slug":"meeting-bot-reinforcement-learning-for","title":"MEETING BOT: Reinforcement Learning for Dialogue Based Meeting Scheduling","date":"2018-12-28","arxiv_id":"1812.11158","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-representation-learning-with-recurrent","title":"State representation learning with recurrent capsule networks","date":"2018-12-28","arxiv_id":"1812.11202","repositories_listed":0,"syntology":null},{"url":null,"slug":"dealing-with-limited-backhaul-capacity-in","title":"Dealing with Limited Backhaul Capacity in Millimeter Wave Systems: A Deep Reinforcement Learning Approach","date":"2018-12-27","arxiv_id":"1901.01119","repositories_listed":0,"syntology":null},{"url":null,"slug":"double-neural-counterfactual-regret","title":"Double Neural Counterfactual Regret Minimization","date":"2018-12-27","arxiv_id":"1812.10607","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-architecture-for","title":"Quantum Adiabatic Algorithm Design using Reinforcement Learning","date":"2018-12-27","arxiv_id":"1812.10797","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-concept-of-deep-reinforcement-learning","title":"A New Concept of Deep Reinforcement Learning based Augmented General Sequence Tagging System","date":"2018-12-26","arxiv_id":"1812.10234","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-refine-source-representations-for","title":"Learning to Refine Source Representations for Neural Machine Translation","date":"2018-12-26","arxiv_id":"1812.10230","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-walk-via-deep-reinforcement","title":"Learning to Walk via Deep Reinforcement Learning","date":"2018-12-26","arxiv_id":"1812.11103","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-market-making-using-multi-agent","title":"Optimizing Market Making using Multi-Agent Reinforcement Learning","date":"2018-12-26","arxiv_id":"1812.10252","repositories_listed":0,"syntology":null},{"url":null,"slug":"moment-matching-training-for-neural-machine","title":"Moment Matching Training for Neural Machine Translation: A Preliminary Study","date":"2018-12-24","arxiv_id":"1812.09836","repositories_listed":0,"syntology":null},{"url":null,"slug":"vmav-c-a-deep-attention-based-reinforcement","title":"VMAV-C: A Deep Attention-based Reinforcement Learning Algorithm for Model-based Control","date":"2018-12-24","arxiv_id":"1812.09968","repositories_listed":0,"syntology":null},{"url":null,"slug":"estimating-rationally-inattentive-utility","title":"Estimating Rationally Inattentive Utility Functions with Deep Clustering for Framing - Applications in YouTube Engagement Dynamics","date":"2018-12-23","arxiv_id":"1812.09640","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallelized-interactive-machine-learning-on","title":"Parallelized Interactive Machine Learning on Autonomous Vehicles","date":"2018-12-23","arxiv_id":"1812.09724","repositories_listed":0,"syntology":null},{"url":null,"slug":"escape-room-a-configurable-testbed-for","title":"Escape Room: A Configurable Testbed for Hierarchical Reinforcement Learning","date":"2018-12-22","arxiv_id":"1812.09521","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-transformation-policy-network-for","title":"Graph Transformation Policy Network for Chemical Reaction Prediction","date":"2018-12-22","arxiv_id":"1812.09441","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-navigate-the-web","title":"Learning to Navigate the Web","date":"2018-12-21","arxiv_id":"1812.09195","repositories_listed":0,"syntology":null},{"url":null,"slug":"nadpex-an-on-policy-temporally-consistent","title":"NADPEx: An on-policy temporally consistent exploration method for deep reinforcement learning","date":"2018-12-21","arxiv_id":"1812.09028","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-quantum-error-correction-codes","title":"Optimizing Quantum Error Correction Codes with Reinforcement Learning","date":"2018-12-20","arxiv_id":"1812.08451","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-online-learning-via-meta-learning","title":"Deep Online Learning via Meta-Learning: Continual Adaptation for Model-Based RL","date":"2018-12-18","arxiv_id":"1812.07671","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-search","title":"Deep reinforcement learning for search, recommendation, and online advertising: a survey","date":"2018-12-18","arxiv_id":"1812.07127","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adaptation-for-reinforcement-learning","title":"Domain Adaptation for Reinforcement Learning on the Atari","date":"2018-12-18","arxiv_id":"1812.07452","repositories_listed":0,"syntology":null},{"url":null,"slug":"incentive-based-demand-response-for-smart","title":"Incentive-based demand response for smart grid with reinforcement learning and deep neural network","date":"2018-12-18","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-to-real-via-sim-to-sim-data-efficient","title":"Sim-to-Real via Sim-to-Sim: Data-efficient Robotic Grasping via Randomized-to-Canonical Adaptation Networks","date":"2018-12-18","arxiv_id":"1812.07252","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-multimodal-model-agnostic-meta","title":"Toward Multimodal Model-Agnostic Meta-Learning","date":"2018-12-18","arxiv_id":"1812.07172","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-meta-reinforcement-learning-for","title":"A Review of Meta-Reinforcement Learning for Deep Neural Networks Architecture Search","date":"2018-12-17","arxiv_id":"1812.07995","repositories_listed":0,"syntology":null},{"url":null,"slug":"fuzzy-controller-of-reward-of-reinforcement","title":"Fuzzy Controller of Reward of Reinforcement Learning For Handwritten Digit Recognition","date":"2018-12-17","arxiv_id":"1812.07028","repositories_listed":0,"syntology":null},{"url":null,"slug":"malthusian-reinforcement-learning","title":"Malthusian Reinforcement Learning","date":"2018-12-17","arxiv_id":"1812.07019","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-adaptive-caching","title":"Reinforcement Learning for Adaptive Caching with Dynamic Storage Pricing","date":"2018-12-17","arxiv_id":"1812.08593","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-logarithmic-barrier-method-for-proximal","title":"A Logarithmic Barrier Method For Proximal Policy Optimization","date":"2018-12-16","arxiv_id":"1812.06502","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-active-information-seeking-model-for-goal","title":"Gold Seeker: Information Gain from Policy Distributions for Goal-oriented Vision-and-Langauge Reasoning","date":"2018-12-16","arxiv_id":"1812.06398","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-likelihood-quantile-networks","title":"Likelihood Quantile Networks for Coordinating Multi-Agent Reinforcement Learning","date":"2018-12-15","arxiv_id":"1812.06319","repositories_listed":0,"syntology":null},{"url":null,"slug":"guaranteed-satisficing-and-finite-regret","title":"Guaranteed satisficing and finite regret: Analysis of a cognitive satisficing value function","date":"2018-12-14","arxiv_id":"1812.05795","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-shared-model-governance-via-model","title":"Scaling shared model governance via model splitting","date":"2018-12-14","arxiv_id":"1812.05979","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-entropy-of-artificial-intelligence-and-a","title":"The Entropy of Artificial Intelligence and a Case Study of AlphaZero from Shannon's Perspective","date":"2018-12-14","arxiv_id":"1812.05794","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-communicate-a-machine-learning","title":"Learning to Communicate: A Machine Learning Framework for Heterogeneous Multi-Agent Robotic Systems","date":"2018-12-13","arxiv_id":"1812.05256","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-exploration-of-nonlinear-dynamical","title":"A predictive safety filter for learning-based control of constrained nonlinear dynamical systems","date":"2018-12-13","arxiv_id":"1812.05506","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-recomposition-by-learning-based-icp","title":"Scene Recomposition by Learning-based ICP","date":"2018-12-13","arxiv_id":"1812.05583","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-neural-networks-algorithms-for-1","title":"Deep neural networks algorithms for stochastic control problems on finite horizon: convergence analysis","date":"2018-12-11","arxiv_id":"1812.04300","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-model-free-reinforcement-learning","title":"Efficient Model-Free Reinforcement Learning Using Gaussian Process","date":"2018-12-11","arxiv_id":"1812.04359","repositories_listed":0,"syntology":null},{"url":null,"slug":"kf-lax-kronecker-factored-curvature","title":"KF-LAX: Kronecker-factored curvature estimation for control variate optimization in reinforcement learning","date":"2018-12-11","arxiv_id":"1812.04181","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-gap-between-model-based-and-model-free","title":"The Gap Between Model-Based and Model-Free Methods on the Linear Quadratic Regulator: An Asymptotic Viewpoint","date":"2018-12-09","arxiv_id":"1812.03565","repositories_listed":0,"syntology":null}],"record_sha256":"5dfcd13880bbd0c57566b1896bdbbb1fce06e592947fd903a2002a6877a3fba7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}