{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/103","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":103,"pages_in_order":132,"rows_per_page":100,"rows":[10201,10300],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/102","next":"/task/reinforcement-learning/papers/104","papers":[{"url":null,"slug":"multi-agent-hierarchical-reinforcement","title":"Multi-agent Hierarchical Reinforcement Learning with Dynamic Termination","date":"2019-10-21","arxiv_id":"1910.09508","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-optimization-for-mathcalh_2-linear","title":"Policy Optimization for $\\mathcal{H}_2$ Linear Control with $\\mathcal{H}_\\infty$ Robustness Guarantee: Implicit Regularization and Global Convergence","date":"2019-10-21","arxiv_id":"1910.09496","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-allocation-in-mobility-aware","title":"Resource Allocation in Mobility-Aware Federated Learning Networks: A Deep Reinforcement Learning Approach","date":"2019-10-21","arxiv_id":"1910.09172","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-educated-language-agent-with-hindsight-1","title":"HIGhER : Improving instruction following with Hindsight Generation for Experience Replay","date":"2019-10-21","arxiv_id":"1910.09451","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-sim-to-real-adaptation-for","title":"Self-Supervised Sim-to-Real Adaptation for Visual Robotic Manipulation","date":"2019-10-21","arxiv_id":"1910.09470","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-industrial-management-via","title":"Autonomous Industrial Management via Reinforcement Learning: Self-Learning Agents for Decision-Making -- A Review","date":"2019-10-20","arxiv_id":"1910.08942","repositories_listed":0,"syntology":null},{"url":null,"slug":"diverse-behavior-is-what-game-ai-needs","title":"Diverse Behavior Is What Game AI Needs: Generating Varied Human-Like Playing Styles Using Evolutionary Multi-Objective Deep Reinforcement Learning","date":"2019-10-20","arxiv_id":"1910.09022","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-6d-multi-object-pose-estimation-in","title":"Active 6D Multi-Object Pose Estimation in Cluttered Scenarios with Deep Reinforcement Learning","date":"2019-10-19","arxiv_id":"1910.08811","repositories_listed":0,"syntology":null},{"url":null,"slug":"explainable-ai-deep-reinforcement-learning","title":"Explainable AI: Deep Reinforcement Learning Agents for Residential Demand Side Cost Savings in Smart Grids","date":"2019-10-19","arxiv_id":"1910.08719","repositories_listed":0,"syntology":null},{"url":null,"slug":"opinion-shaping-in-social-networks-using","title":"Opinion shaping in social networks using reinforcement learning","date":"2019-10-19","arxiv_id":"1910.08802","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-convolutional-policy-for-solving-tree","title":"Graph Convolutional Policy for Solving Tree Decomposition via Reinforcement Learning Heuristics","date":"2019-10-18","arxiv_id":"1910.08371","repositories_listed":0,"syntology":null},{"url":null,"slug":"offworld-gym-open-access-physical-robotics","title":"OffWorld Gym: open-access physical robotics environment for real-world reinforcement learning benchmark and research","date":"2019-10-18","arxiv_id":"1910.08639","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-connections-between-constrained","title":"On Connections between Constrained Optimization and Reinforcement Learning","date":"2019-10-18","arxiv_id":"1910.08476","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-sample-complexity-of-actor-critic","title":"On the Sample Complexity of Actor-Critic Method for Reinforcement Learning with Function Approximation","date":"2019-10-18","arxiv_id":"1910.08412","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-context-rewriting-for-open","title":"Unsupervised Context Rewriting for Open Domain Conversation","date":"2019-10-18","arxiv_id":"1910.08282","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-deep-learning-techniques-for-1","title":"A Survey of Deep Learning Techniques for Autonomous Driving","date":"2019-10-17","arxiv_id":"1910.07738","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-provably-finds-nash-equilibria","title":"Actor-Critic Provably Finds Nash Equilibria of Linear-Quadratic Mean-Field Games","date":"2019-10-16","arxiv_id":"1910.07498","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-trade-offs-in-off-policy-learning","title":"Adaptive Trade-Offs in Off-Policy Learning","date":"2019-10-16","arxiv_id":"1910.07478","repositories_listed":0,"syntology":null},{"url":null,"slug":"conditional-importance-sampling-for-off","title":"Conditional Importance Sampling for Off-Policy Learning","date":"2019-10-16","arxiv_id":"1910.07479","repositories_listed":0,"syntology":null},{"url":null,"slug":"creativity-in-robot-manipulation-with-deep","title":"Creativity in Robot Manipulation with Deep Reinforcement Learning","date":"2019-10-16","arxiv_id":"1910.07459","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-agnostic-meta-learning-using-runge","title":"Model-Agnostic Meta-Learning using Runge-Kutta Methods","date":"2019-10-16","arxiv_id":"1910.07368","repositories_listed":0,"syntology":null},{"url":null,"slug":"negatively-correlated-search-as-a-parallel","title":"Parallel Exploration via Negatively Correlated Search","date":"2019-10-16","arxiv_id":"1910.07151","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-bit-allocation-under-task-driven","title":"Reinforced Bit Allocation under Task-Driven Semantic Distortion Metrics","date":"2019-10-16","arxiv_id":"1910.07392","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-a-minimal-learning-agent-can-infer-the","title":"How a minimal learning agent can infer the existence of unobserved variables in a complex environment","date":"2019-10-15","arxiv_id":"1910.06985","repositories_listed":0,"syntology":null},{"url":null,"slug":"safecritic-collision-aware-trajectory","title":"SafeCritic: Collision-Aware Trajectory Prediction","date":"2019-10-15","arxiv_id":"1910.06673","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-the-curse-of-horizon-in-off","title":"Understanding the Curse of Horizon in Off-Policy Evaluation via Conditional Importance Sampling","date":"2019-10-15","arxiv_id":"1910.06508","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-unified-view-of-likelihood-ratio-and","title":"A unified view of likelihood ratio and reparameterization gradients and an optimal importance sampling scheme","date":"2019-10-14","arxiv_id":"1910.06419","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-with-differentially-private","title":"Actor Critic with Differentially Private Critic","date":"2019-10-14","arxiv_id":"1910.05876","repositories_listed":0,"syntology":null},{"url":null,"slug":"coordination-of-pv-smart-inverters-using-deep","title":"Coordination of PV Smart Inverters Using Deep Reinforcement Learning for Grid Voltage Regulation","date":"2019-10-14","arxiv_id":"1910.05907","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-graph-configuration-with","title":"Dynamic Graph Configuration with Reinforcement Learning for Connected Autonomous Vehicle Trajectories","date":"2019-10-14","arxiv_id":"1910.06788","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-transfer-reinforcement-learning-for","title":"Federated Transfer Reinforcement Learning for Autonomous Driving","date":"2019-10-14","arxiv_id":"1910.06001","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-reduction-of-variance-and","title":"On the Reduction of Variance and Overestimation of Deep Q-Learning","date":"2019-10-14","arxiv_id":"1910.05983","repositories_listed":0,"syntology":null},{"url":null,"slug":"extracting-incentives-from-black-box","title":"Extracting Incentives from Black-Box Decisions","date":"2019-10-13","arxiv_id":"1910.05664","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-program-synthesis-by-self-learning-1","title":"Neural Program Synthesis By Self-Learning","date":"2019-10-13","arxiv_id":"1910.05865","repositories_listed":0,"syntology":null},{"url":null,"slug":"qos-and-jamming-aware-wireless-networking","title":"QoS and Jamming-Aware Wireless Networking Using Deep Reinforcement Learning","date":"2019-10-13","arxiv_id":"1910.05766","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-exposure-bias-in-language-modeling","title":"Rethinking Exposure Bias In Language Modeling","date":"2019-10-13","arxiv_id":"1910.11235","repositories_listed":0,"syntology":null},{"url":null,"slug":"curiosity-driven-recommendation-strategy-for","title":"Curiosity-Driven Recommendation Strategy for Adaptive Learning via Deep Reinforcement Learning","date":"2019-10-12","arxiv_id":"1910.12577","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-inference-and-exploration-for","title":"Uncertainty Quantification and Exploration for Reinforcement Learning","date":"2019-10-12","arxiv_id":"1910.05471","repositories_listed":0,"syntology":null},{"url":null,"slug":"evaluating-disentangled-representations","title":"How to Not Measure Disentanglement","date":"2019-10-12","arxiv_id":"1910.05587","repositories_listed":0,"syntology":null},{"url":null,"slug":"regularizing-model-based-planning-with-energy","title":"Regularizing Model-Based Planning with Energy-Based Models","date":"2019-10-12","arxiv_id":"1910.05527","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-optimization-meets-riemannian","title":"Bayesian Optimization Meets Riemannian Manifolds in Robot Learning","date":"2019-10-11","arxiv_id":"1910.04998","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-hvac-scheduling-using-reinforcement","title":"Building HVAC Scheduling Using Reinforcement Learning via Neural Network Based Model Approximation","date":"2019-10-11","arxiv_id":"1910.05313","repositories_listed":0,"syntology":null},{"url":null,"slug":"green-deep-reinforcement-learning-for-radio","title":"Green Deep Reinforcement Learning for Radio Resource Management: Architecture, Algorithm Compression and Challenge","date":"2019-10-11","arxiv_id":"1910.05054","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-gradient-estimation-in-evolutionary-1","title":"Improving Gradient Estimation in Evolutionary Strategies With Past Descent Directions","date":"2019-10-11","arxiv_id":"1910.05268","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-cyber-physical-human-systems-via-an","title":"Modeling Cyber-Physical Human Systems via an Interplay Between Reinforcement Learning and Game Theory","date":"2019-10-11","arxiv_id":"1910.05092","repositories_listed":0,"syntology":null},{"url":null,"slug":"zap-q-learning-with-nonlinear-function","title":"Zap Q-Learning With Nonlinear Function Approximation","date":"2019-10-11","arxiv_id":"1910.05405","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-driving-using-safe-reinforcement","title":"Autonomous Driving using Safe Reinforcement Learning by Incorporating a Regret-based Human Lane-Changing Decision Model","date":"2019-10-10","arxiv_id":"1910.04803","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-intrinsically-motivated-robotic","title":"Efficient Intrinsically Motivated Robotic Grasping with Learning-Adaptive Imagination in Latent Space","date":"2019-10-10","arxiv_id":"1910.04729","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-dual-hormone-closed-loop-delivery-system","title":"A Dual-Hormone Closed-Loop Delivery System for Type 1 Diabetes Using Deep Reinforcement Learning","date":"2019-10-09","arxiv_id":"1910.04059","repositories_listed":0,"syntology":null},{"url":null,"slug":"compatible-features-for-monotonic-policy","title":"Compatible features for Monotonic Policy Improvement","date":"2019-10-09","arxiv_id":"1910.03880","repositories_listed":0,"syntology":null},{"url":null,"slug":"ctrl-z-recovering-from-instability-in","title":"Ctrl-Z: Recovering from Instability in Reinforcement Learning","date":"2019-10-09","arxiv_id":"1910.03732","repositories_listed":0,"syntology":null},{"url":null,"slug":"defensive-escort-teams-via-multi-agent-deep","title":"Defensive Escort Teams via Multi-Agent Deep Reinforcement Learning","date":"2019-10-09","arxiv_id":"1910.04537","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-task-adaptation-for-tasks-labeled-using","title":"Fast Task-Adaptation for Tasks Labeled Using Natural Language in Reinforcement Learning","date":"2019-10-09","arxiv_id":"1910.04040","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-deep-double-q-routing","title":"Hierarchical Deep Double Q-Routing","date":"2019-10-09","arxiv_id":"1910.04041","repositories_listed":0,"syntology":null},{"url":null,"slug":"imagined-value-gradients-model-based-policy","title":"Imagined Value Gradients: Model-Based Policy Optimization with Transferable Latent Dynamics Models","date":"2019-10-09","arxiv_id":"1910.04142","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-generalization-in-meta-1","title":"Improving Generalization in Meta Reinforcement Learning using Learned Objectives","date":"2019-10-09","arxiv_id":"1910.04098","repositories_listed":0,"syntology":null},{"url":null,"slug":"integrating-behavior-cloning-and","title":"Integrating Behavior Cloning and Reinforcement Learning for Improved Performance in Dense and Sparse Reward Environments","date":"2019-10-09","arxiv_id":"1910.04281","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-induced-deep-q-network-for-a-slide","title":"Learning Visual Affordances with Target-Orientated Deep Q-Network to Grasp Objects by Harnessing Environmental Fixtures","date":"2019-10-09","arxiv_id":"1910.03781","repositories_listed":0,"syntology":null},{"url":null,"slug":"linear-quadratic-mean-field-reinforcement","title":"Linear-Quadratic Mean-Field Reinforcement Learning: Convergence of Policy Gradient Methods","date":"2019-10-09","arxiv_id":"1910.04295","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-exploiting","title":"Model-Based Reinforcement Learning Exploiting State-Action Equivalence","date":"2019-10-09","arxiv_id":"1910.04077","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-based-reinforcement-learning-for-1","title":"Model-based Reinforcement Learning for Predictions and Control for Limit Order Books","date":"2019-10-09","arxiv_id":"1910.03743","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiple-objective-reinforcement-learning-for","title":"Multiple-objective Reinforcement Learning for Inverse Design and Identification","date":"2019-10-09","arxiv_id":"1910.03741","repositories_listed":0,"syntology":null},{"url":null,"slug":"stochastic-implicit-natural-gradient-for","title":"Black-box Optimizer with Implicit Natural Gradient","date":"2019-10-09","arxiv_id":"1910.04301","repositories_listed":0,"syntology":null},{"url":null,"slug":"tactical-reward-shaping-bypassing","title":"Tactical Reward Shaping: Bypassing Reinforcement Learning with Strategy-Based Goals","date":"2019-10-08","arxiv_id":"1910.03144","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-a-good-representation-sufficient-for","title":"Is a Good Representation Sufficient for Sample Efficient Reinforcement Learning?","date":"2019-10-07","arxiv_id":"1910.03016","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-reinforcement-learning-for-order","title":"Multi-Agent Reinforcement Learning for Order-dispatching via Order-Vehicle Distribution Matching","date":"2019-10-07","arxiv_id":"1910.02591","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-step-greedy-policies-in-model-free-deep-1","title":"Multi-step Greedy Reinforcement Learning Algorithms","date":"2019-10-07","arxiv_id":"1910.02919","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-with-structured-1","title":"Reinforcement Learning with Structured Hierarchical Grammar Representations of Actions","date":"2019-10-07","arxiv_id":"1910.02876","repositories_listed":0,"syntology":null},{"url":null,"slug":"biased-aggregation-rollout-and-enhanced","title":"Biased Aggregation, Rollout, and Enhanced Policy Improvement for Reinforcement Learning","date":"2019-10-06","arxiv_id":"1910.02426","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimising-energy-and-overhead-for-large","title":"Optimising energy and overhead for large parameter space simulations","date":"2019-10-06","arxiv_id":"1910.02516","repositories_listed":0,"syntology":null},{"url":null,"slug":"probabilistic-successor-representations-with","title":"Probabilistic Successor Representations with Kalman Temporal Differences","date":"2019-10-06","arxiv_id":"1910.02532","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-based-fault-tolerant-approach-for","title":"Attention-based Fault-tolerant Approach for Multi-agent Reinforcement Learning Systems","date":"2019-10-05","arxiv_id":"1910.02240","repositories_listed":0,"syntology":null},{"url":null,"slug":"deepmnavigate-deep-reinforced-multi-robot","title":"DeepMNavigate: Deep Reinforced Multi-Robot Navigation Unifying Local & Global Collision Avoidance","date":"2019-10-04","arxiv_id":"1910.09441","repositories_listed":0,"syntology":null},{"url":null,"slug":"discounted-reinforcement-learning-is-not-an","title":"Discounted Reinforcement Learning Is Not an Optimization Problem","date":"2019-10-04","arxiv_id":"1910.02140","repositories_listed":0,"syntology":null},{"url":null,"slug":"if-maxent-rl-is-the-answer-what-is-the","title":"If MaxEnt RL is the Answer, What is the Question?","date":"2019-10-04","arxiv_id":"1910.01913","repositories_listed":0,"syntology":null},{"url":null,"slug":"im-sorry-dave-im-afraid-i-cant-do-that-deep-q","title":"I'm sorry Dave, I'm afraid I can't do that, Deep Q-learning from forbidden action","date":"2019-10-04","arxiv_id":"1910.02078","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-robust-representations-with-graph","title":"Learning Robust Representations with Graph Denoising Policy Network","date":"2019-10-04","arxiv_id":"1910.01784","repositories_listed":0,"syntology":null},{"url":null,"slug":"manufacturing-dispatching-using-reinforcement","title":"Manufacturing Dispatching using Reinforcement and Transfer Learning","date":"2019-10-04","arxiv_id":"1910.02035","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-learning-on-simulated-robots","title":"Zero Shot Learning on Simulated Robots","date":"2019-10-04","arxiv_id":"1910.01994","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-zero-dynamics-inspired-feedback","title":"Hybrid Zero Dynamics Inspired Feedback Control Policy Design for 3D Bipedal Locomotion using Reinforcement Learning","date":"2019-10-03","arxiv_id":"1910.01748","repositories_listed":0,"syntology":null},{"url":null,"slug":"path-planning-microswimmers-can-swim","title":"Machine learning strategies for path-planning microswimmers in turbulent flows","date":"2019-10-03","arxiv_id":"1910.01728","repositories_listed":0,"syntology":null},{"url":null,"slug":"sensordrop-a-reinforcement-learning-framework","title":"SensorDrop: A Reinforcement Learning Framework for Communication Overhead Reduction on the Edge","date":"2019-10-03","arxiv_id":"1910.01601","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-logical-specifications-of-objectives-in","title":"Using Logical Specifications of Objectives in Multi-Objective Reinforcement Learning","date":"2019-10-03","arxiv_id":"1910.01723","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-assisted-annotator-using-reinforcement","title":"AI Assisted Annotator using Reinforcement Learning","date":"2019-10-02","arxiv_id":"1910.02052","repositories_listed":0,"syntology":null},{"url":null,"slug":"boosting-image-recognition-with-non","title":"Boosting Image Recognition with Non-differentiable Constraints","date":"2019-10-02","arxiv_id":"1910.00736","repositories_listed":0,"syntology":null},{"url":null,"slug":"cwae-irl-formulating-a-supervised-approach-to-1","title":"CWAE-IRL: Formulating a supervised approach to Inverse Reinforcement Learning problem","date":"2019-10-02","arxiv_id":"1910.00584","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-single-shot","title":"Deep Reinforcement Learning for Single-Shot Diagnosis and Adaptation in Damaged Robots","date":"2019-10-02","arxiv_id":"1910.01240","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-language-state-representation-for","title":"Language is Power: Representing States Using Natural Language in Reinforcement Learning","date":"2019-10-02","arxiv_id":"1910.02789","repositories_listed":0,"syntology":null},{"url":null,"slug":"relationship-explainable-multi-objective-1","title":"Relationship Explainable Multi-objective Optimization Via Vector Value Function Based Reinforcement Learning","date":"2019-10-02","arxiv_id":"1910.01919","repositories_listed":0,"syntology":null},{"url":null,"slug":"stabilizing-off-policy-reinforcement-learning","title":"Never Worse, Mostly Better: Stable Policy Improvement in Deep Reinforcement Learning","date":"2019-10-02","arxiv_id":"1910.01062","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-active-learning-for-human","title":"Deep Reinforcement Active Learning for Human-in-the-Loop Person Re-Identification","date":"2019-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fair-loss-margin-aware-reinforcement-learning","title":"Fair Loss: Margin-Aware Reinforcement Learning for Deep Face Recognition","date":"2019-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"generalization-in-generation-a-closer-look-at","title":"Generalization in Generation: A closer look at Exposure Bias","date":"2019-10-01","arxiv_id":"1910.00292","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-paraphrases-with-lean-vocabulary","title":"Generating Paraphrases with Lean Vocabulary","date":"2019-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"machine-translation-for-machines-the","title":"Machine Translation for Machines: the Sentiment Classification Use Case","date":"2019-10-01","arxiv_id":"1910.00478","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantile-qt-opt-for-risk-aware-vision-based","title":"Quantile QT-Opt for Risk-Aware Vision-Based Robotic Grasping","date":"2019-10-01","arxiv_id":"1910.02787","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-multi-objective","title":"Reinforcement Learning for Multi-Objective Optimization of Online Decisions in High-Dimensional Systems","date":"2019-10-01","arxiv_id":"1910.00211","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-critical-attention-learning-for-person","title":"Self-Critical Attention Learning for Person Re-Identification","date":"2019-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-interaction-aware-scene-understanding","title":"Dynamic Interaction-Aware Scene Understanding for Reinforcement Learning in Autonomous Driving","date":"2019-09-30","arxiv_id":"1909.13582","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-meta-reinforcement-learning-via-1","title":"MGHRL: Meta Goal-generation for Hierarchical Reinforcement Learning","date":"2019-09-30","arxiv_id":"1909.13607","repositories_listed":0,"syntology":null}],"record_sha256":"e61ad31fe291a5d5100a40e94e295a8f542b1ef251136d4a313afefc86b84c32","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}