{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/55","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":55,"pages_in_order":59,"rows_per_page":100,"rows":[5401,5500],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/54","next":"/task/deep-reinforcement-learning/papers/56","papers":[{"url":null,"slug":"learning-to-walk-via-deep-reinforcement","title":"Learning to Walk via Deep Reinforcement Learning","date":"2018-12-26","arxiv_id":"1812.11103","repositories_listed":0,"syntology":null},{"url":null,"slug":"parallelized-interactive-machine-learning-on","title":"Parallelized Interactive Machine Learning on Autonomous Vehicles","date":"2018-12-23","arxiv_id":"1812.09724","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-navigate-the-web","title":"Learning to Navigate the Web","date":"2018-12-21","arxiv_id":"1812.09195","repositories_listed":0,"syntology":null},{"url":null,"slug":"nadpex-an-on-policy-temporally-consistent","title":"NADPEx: An on-policy temporally consistent exploration method for deep reinforcement learning","date":"2018-12-21","arxiv_id":"1812.09028","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-search","title":"Deep reinforcement learning for search, recommendation, and online advertising: a survey","date":"2018-12-18","arxiv_id":"1812.07127","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-adaptation-for-reinforcement-learning","title":"Domain Adaptation for Reinforcement Learning on the Atari","date":"2018-12-18","arxiv_id":"1812.07452","repositories_listed":0,"syntology":null},{"url":null,"slug":"scene-recomposition-by-learning-based-icp","title":"Scene Recomposition by Learning-based ICP","date":"2018-12-13","arxiv_id":"1812.05583","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-neural-networks-algorithms-for-1","title":"Deep neural networks algorithms for stochastic control problems on finite horizon: convergence analysis","date":"2018-12-11","arxiv_id":"1812.04300","repositories_listed":0,"syntology":null},{"url":null,"slug":"measuring-and-characterizing-generalization","title":"Measuring and Characterizing Generalization in Deep Reinforcement Learning","date":"2018-12-07","arxiv_id":"1812.02868","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-deep-reinforcement-learning-driving","title":"Zero-shot Deep Reinforcement Learning Driving Policy Transfer for Autonomous Vehicles based on Robust Control","date":"2018-12-07","arxiv_id":"1812.03216","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-and-the-deadly","title":"Deep Reinforcement Learning and the Deadly Triad","date":"2018-12-06","arxiv_id":"1812.02648","repositories_listed":0,"syntology":null},{"url":null,"slug":"bach2bach-generating-music-using-a-deep","title":"Bach2Bach: Generating Music Using A Deep Reinforcement Learning Approach","date":"2018-12-03","arxiv_id":"1812.01060","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-intelligent","title":"Deep Reinforcement Learning for Intelligent Transportation Systems","date":"2018-12-03","arxiv_id":"1812.00979","repositories_listed":0,"syntology":null},{"url":null,"slug":"foldingzero-protein-folding-from-scratch-in","title":"FoldingZero: Protein Folding from Scratch in Hydrophobic-Polar Model","date":"2018-12-03","arxiv_id":"1812.00967","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-diverse-programs-with-instruction","title":"Generating Diverse Programs with Instruction Conditioned Reinforced Adversarial Learning","date":"2018-12-03","arxiv_id":"1812.00898","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-with","title":"Multi-agent Deep Reinforcement Learning with Extremely Noisy Observations","date":"2018-12-03","arxiv_id":"1812.00922","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-constrained-deep-reinforcement","title":"Resource Constrained Deep Reinforcement Learning","date":"2018-12-03","arxiv_id":"1812.00600","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-measurement-scheduling-for-adverse","title":"Dynamic Measurement Scheduling for Adverse Event Forecasting using Deep RL","date":"2018-12-01","arxiv_id":"1812.00268","repositories_listed":0,"syntology":null},{"url":null,"slug":"genetic-gated-networks-for-deep-reinforcement-1","title":"Genetic-Gated Networks for Deep Reinforcement Learning","date":"2018-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"refuel-exploring-sparse-features-in-deep","title":"REFUEL: Exploring Sparse Features in Deep Reinforcement Learning for Fast Disease Diagnosis","date":"2018-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-organize-your-deep-reinforcement","title":"How to Organize your Deep Reinforcement Learning Agents: The Importance of Communication Topology","date":"2018-11-30","arxiv_id":"1811.12556","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-monte-carlo-tree-search-as-a","title":"Using Monte Carlo Tree Search as a Demonstrator within Asynchronous Deep RL","date":"2018-11-30","arxiv_id":"1812.00045","repositories_listed":0,"syntology":null},{"url":null,"slug":"flow-shape-design-for-microfluidic-devices","title":"Flow Shape Design for Microfluidic Devices Using Deep Reinforcement Learning","date":"2018-11-29","arxiv_id":"1811.12444","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-time-optimal","title":"Deep Reinforcement Learning for Time Optimal Velocity Control using Prior Knowledge","date":"2018-11-28","arxiv_id":"1811.11615","repositories_listed":0,"syntology":null},{"url":null,"slug":"trajectory-based-learning-for-ball-in-maze","title":"Trajectory-based Learning for Ball-in-Maze Games","date":"2018-11-28","arxiv_id":"1811.11441","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-face-aging-in-videos-via-deep","title":"Automatic Face Aging in Videos via Deep Reinforcement Learning","date":"2018-11-27","arxiv_id":"1811.11082","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-traffic-light-control-at","title":"Distributed traffic light control at uncoupled intersections with real-world topology by deep reinforcement learning","date":"2018-11-27","arxiv_id":"1811.11233","repositories_listed":0,"syntology":null},{"url":null,"slug":"prioritizing-starting-states-for","title":"Exploring Restart Distributions","date":"2018-11-27","arxiv_id":"1811.11298","repositories_listed":0,"syntology":null},{"url":null,"slug":"quality-aware-multimodal-saliency-detection","title":"Quality-Aware Multimodal Saliency Detection via Deep Reinforcement Learning","date":"2018-11-27","arxiv_id":"1811.10763","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimization-of-information-seeking-dialogue","title":"Optimization of Information-Seeking Dialogue Strategy for Argumentation-Based Dialogue System","date":"2018-11-26","arxiv_id":"1811.10728","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-neural-hybrid-agents-for","title":"Evolutionary-Neural Hybrid Agents for Architecture Search","date":"2018-11-24","arxiv_id":"1811.09828","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-activate-relay-nodes-deep","title":"Learning to Activate Relay Nodes: Deep Reinforcement Learning Approach","date":"2018-11-24","arxiv_id":"1811.09759","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-attend-in-a-brain-inspired-deep","title":"Learning to attend in a brain-inspired deep neural network","date":"2018-11-23","arxiv_id":"1811.09699","repositories_listed":0,"syntology":null},{"url":null,"slug":"simulated-autonomous-driving-in-a-realistic","title":"Simulated Autonomous Driving in a Realistic Driving Environment using Deep Reinforcement Learning and a Deterministic Finite State Machine","date":"2018-11-19","arxiv_id":"1811.07868","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-where-to-fixate-on-foveated-images","title":"Cost-Aware Fine-Grained Recognition for IoTs Based on Sequential Fixations","date":"2018-11-16","arxiv_id":"1811.06868","repositories_listed":0,"syntology":null},{"url":null,"slug":"concept-learning-through-deep-reinforcement","title":"Concept Learning through Deep Reinforcement Learning with Memory-Augmented Neural Networks","date":"2018-11-15","arxiv_id":"1811.06145","repositories_listed":0,"syntology":null},{"url":null,"slug":"orthogonal-policy-gradient-and-autonomous","title":"Orthogonal Policy Gradient and Autonomous Driving Application","date":"2018-11-15","arxiv_id":"1811.06151","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigating-assistance-system-for-quadcopter","title":"Navigating Assistance System for Quadcopter with Deep Reinforcement Learning","date":"2018-11-12","arxiv_id":"1811.04584","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-initial-attempt-of-combining-visual","title":"An initial attempt of combining visual selective attention with deep reinforcement learning","date":"2018-11-11","arxiv_id":"1811.04407","repositories_listed":0,"syntology":null},{"url":null,"slug":"correlation-filter-selection-for-visual","title":"Correlation Filter Selection for Visual Tracking Using Reinforcement Learning","date":"2018-11-08","arxiv_id":"1811.03196","repositories_listed":0,"syntology":null},{"url":null,"slug":"modular-architecture-for-starcraft-ii-with","title":"Modular Architecture for StarCraft II with Deep Reinforcement Learning","date":"2018-11-08","arxiv_id":"1811.03555","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-green","title":"Deep Reinforcement Learning for Green Security Games with Real-Time Information","date":"2018-11-06","arxiv_id":"1811.02483","repositories_listed":0,"syntology":null},{"url":null,"slug":"quasi-newton-optimization-in-deep-q-learning","title":"Deep Reinforcement Learning via L-BFGS Optimization","date":"2018-11-06","arxiv_id":"1811.02693","repositories_listed":0,"syntology":null},{"url":null,"slug":"managing-engineering-systems-with-large-state","title":"Managing engineering systems with large state and action spaces through deep reinforcement learning","date":"2018-11-05","arxiv_id":"1811.02052","repositories_listed":0,"syntology":null},{"url":null,"slug":"releq-an-automatic-reinforcement-learning","title":"ReLeQ: A Reinforcement Learning Approach for Deep Quantization of Neural Networks","date":"2018-11-05","arxiv_id":"1811.01704","repositories_listed":0,"syntology":null},{"url":null,"slug":"automated-theorem-proving-in-intuitionistic","title":"Automated Theorem Proving in Intuitionistic Propositional Logic by Deep Reinforcement Learning","date":"2018-11-02","arxiv_id":"1811.00796","repositories_listed":0,"syntology":null},{"url":null,"slug":"macquarie-university-at-bioasq-6b-deep","title":"Macquarie University at BioASQ 6b: Deep learning and deep reinforcement learning for query-based summarisation","date":"2018-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sdrl-interpretable-and-data-efficient-deep","title":"SDRL: Interpretable and Data-efficient Deep Reinforcement Learning Leveraging Symbolic Planning","date":"2018-10-31","arxiv_id":"1811.00090","repositories_listed":0,"syntology":null},{"url":null,"slug":"relative-importance-sampling-for-off-policy","title":"Relative Importance Sampling for off-Policy Actor-Critic in Deep Reinforcement Learning","date":"2018-10-30","arxiv_id":"1810.12558","repositories_listed":0,"syntology":null},{"url":null,"slug":"social-vehicle-swarms-a-novel-perspective-on","title":"Social Vehicle Swarms: A Novel Perspective on Social-aware Vehicular Communication Architecture","date":"2018-10-29","arxiv_id":"1810.11947","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributive-dynamic-spectrum-access-through","title":"Distributive Dynamic Spectrum Access through Deep Reinforcement Learning: A Reservoir Computing Based Approach","date":"2018-10-28","arxiv_id":"1810.11758","repositories_listed":0,"syntology":null},{"url":null,"slug":"differential-variable-speed-limits-control","title":"Differential Variable Speed Limits Control for Freeway Recurrent Bottlenecks via Deep Reinforcement learning","date":"2018-10-25","arxiv_id":"1810.10952","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-modeling-game-for-deriving-theoretical","title":"Meta-modeling game for deriving theoretical-consistent, micro-structural-based traction-separation laws via deep reinforcement learning","date":"2018-10-24","arxiv_id":"1810.10535","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-approaches-for-reinforcement","title":"Hierarchical Approaches for Reinforcement Learning in Parameterized Action Space","date":"2018-10-23","arxiv_id":"1810.09656","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-faults-in-our-pi-stars-security-issues","title":"The Faults in Our Pi Stars: Security Issues and Open Challenges in Deep Reinforcement Learning","date":"2018-10-23","arxiv_id":"1810.10369","repositories_listed":0,"syntology":null},{"url":null,"slug":"transfer-learning-versus-multi-agent-learning","title":"Transfer Learning versus Multi-agent Learning regarding Distributed Decision-Making in Highway Traffic","date":"2018-10-19","arxiv_id":"1810.08515","repositories_listed":0,"syntology":null},{"url":null,"slug":"applications-of-deep-reinforcement-learning","title":"Applications of Deep Reinforcement Learning in Communications and Networking: A Survey","date":"2018-10-18","arxiv_id":"1810.07862","repositories_listed":0,"syntology":null},{"url":null,"slug":"at-human-speed-deep-reinforcement-learning","title":"At Human Speed: Deep Reinforcement Learning with Action Delay","date":"2018-10-16","arxiv_id":"1810.07286","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-deep-reinforcement-learning-for-the","title":"Using Deep Reinforcement Learning for the Continuous Control of Robotic Arms","date":"2018-10-15","arxiv_id":"1810.06746","repositories_listed":0,"syntology":null},{"url":null,"slug":"dexterous-manipulation-with-deep","title":"Dexterous Manipulation with Deep Reinforcement Learning: Efficient, General, and Low-Cost","date":"2018-10-14","arxiv_id":"1810.06045","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-multiagent-deep-reinforcement-learning-the","title":"A Survey and Critique of Multiagent Deep Reinforcement Learning","date":"2018-10-12","arxiv_id":"1810.05587","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-design-for-active-sequential","title":"Policy Design for Active Sequential Hypothesis Testing using Deep Learning","date":"2018-10-11","arxiv_id":"1810.04859","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-wildfire-surveillance-with","title":"Distributed Wildfire Surveillance with Autonomous Aircraft using Deep Reinforcement Learning","date":"2018-10-09","arxiv_id":"1810.04244","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-cognitive-smart-cities-using-big","title":"Enabling Cognitive Smart Cities Using Big Data and Machine Learning: Approaches and Challenges","date":"2018-10-09","arxiv_id":"1810.04107","repositories_listed":0,"syntology":null},{"url":null,"slug":"realizing-learned-quadruped-locomotion","title":"Realizing Learned Quadruped Locomotion Behaviors through Kinematic Motion Primitives","date":"2018-10-09","arxiv_id":"1810.03842","repositories_listed":0,"syntology":null},{"url":null,"slug":"actor-critic-deep-reinforcement-learning-for","title":"Actor-Critic Deep Reinforcement Learning for Dynamic Multichannel Access","date":"2018-10-08","arxiv_id":"1810.03695","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for","title":"Multi-agent Deep Reinforcement Learning for Zero Energy Communities","date":"2018-10-08","arxiv_id":"1810.03679","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-time","title":"Deep Reinforcement Learning for Time Scheduling in RF-Powered Backscatter Cognitive Radio Networks","date":"2018-10-03","arxiv_id":"1810.04520","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-sub-domain-modeling-for-dialogue","title":"Autonomous Sub-domain Modeling for Dialogue Policy with Hierarchical Deep Reinforcement Learning","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-learning-based-on-reward","title":"Curriculum Learning Based on Reward Sparseness for Deep Reinforcement Learning of Task Completion Dialogue Management","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-and-planning-with-a-semantic-model","title":"Learning and Planning with a Semantic Model","date":"2018-09-28","arxiv_id":"1809.10842","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-multiagent-reinforcement","title":"COLLABORATIVE MULTIAGENT REINFORCEMENT LEARNING IN HOMOGENEOUS SWARMS","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"constraining-action-sequences-with-formal","title":"Constraining Action Sequences with Formal Languages for Deep Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-of-universal","title":"Deep Reinforcement Learning of Universal Policies with Diverse Environment Summaries","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-pricing-on-e-commerce-platform-with-1","title":"Dynamic Pricing on E-commerce Platform with Deep Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"expressiveness-in-deep-reinforcement-learning","title":"Expressiveness in Deep Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"guided-exploration-in-deep-reinforcement","title":"Guided Exploration in Deep Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-physics-priors-for-deep","title":"Learning Physics Priors for Deep Reinforcement Learing","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"shrinkage-based-bias-variance-trade-off-for","title":"Shrinkage-based Bias-Variance Trade-off for Deep Reinforcement Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-wisdom-of-the-crowd-reliable-deep","title":"The wisdom of the crowd: reliable deep reinforcement learning through ensembles of Q-functions","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-more-theoretically-grounded-particle","title":"Towards More Theoretically-Grounded Particle Optimization Sampling for Deep Learning","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"where-off-policy-deep-reinforcement-learning","title":"Where Off-Policy Deep Reinforcement Learning Fails","date":"2018-09-27","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"alphaseq-sequence-discovery-with-deep","title":"AlphaSeq: Sequence Discovery with Deep Reinforcement Learning","date":"2018-09-26","arxiv_id":"1810.01218","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-navigation-behaviors-end-to-end-with","title":"Learning Navigation Behaviors End-to-End with AutoRL","date":"2018-09-26","arxiv_id":"1809.10124","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-through-probing-a-decentralized","title":"Learning through Probing: a decentralized reinforcement learning architecture for social dilemmas","date":"2018-09-26","arxiv_id":"1809.10007","repositories_listed":0,"syntology":null},{"url":null,"slug":"anderson-acceleration-for-reinforcement","title":"Anderson Acceleration for Reinforcement Learning","date":"2018-09-25","arxiv_id":"1809.09501","repositories_listed":0,"syntology":null},{"url":null,"slug":"sdn-flow-entry-management-using-reinforcement","title":"SDN Flow Entry Management Using Reinforcement Learning","date":"2018-09-24","arxiv_id":"1809.09003","repositories_listed":0,"syntology":null},{"url":null,"slug":"geometric-multi-model-fitting-by-deep","title":"Geometric Multi-Model Fitting by Deep Reinforcement Learning","date":"2018-09-22","arxiv_id":"1809.08397","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-sim-to-real-transfer-with-modular","title":"Sim-to-Real Transfer of Robot Learning with Variable Length Inputs","date":"2018-09-20","arxiv_id":"1809.07480","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-edge-ai-intelligentizing-mobile-edge","title":"In-Edge AI: Intelligentizing Mobile Edge Computing, Caching and Communication by Federated Learning","date":"2018-09-19","arxiv_id":"1809.07857","repositories_listed":0,"syntology":null},{"url":null,"slug":"switching-isotropic-and-directional","title":"Switching Isotropic and Directional Exploration with Parameter Space Noise in Deep Reinforcement Learning","date":"2018-09-18","arxiv_id":"1809.06570","repositories_listed":0,"syntology":null},{"url":null,"slug":"curriculum-goal-masking-for-continuous-deep","title":"Curriculum goal masking for continuous deep reinforcement learning","date":"2018-09-17","arxiv_id":"1809.06146","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-sensitive-deep-reinforcement-learning","title":"Object-sensitive Deep Reinforcement Learning","date":"2018-09-17","arxiv_id":"1809.06064","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-reinforcement-learning-for","title":"Adversarial Reinforcement Learning for Observer Design in Autonomous Systems under Cyber Attacks","date":"2018-09-15","arxiv_id":"1809.06784","repositories_listed":0,"syntology":null},{"url":null,"slug":"macquarie-university-at-bioasq-6b-deep-1","title":"Macquarie University at BioASQ 6b: Deep learning and deep reinforcement learning for query-based multi-document summarisation","date":"2018-09-14","arxiv_id":"1809.05283","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-diagnostics-for-deep-reinforcement","title":"Visual Diagnostics for Deep Reinforcement Learning Policy Development","date":"2018-09-14","arxiv_id":"1809.06781","repositories_listed":0,"syntology":null},{"url":null,"slug":"image-captioning-based-on-deep-reinforcement","title":"Image Captioning based on Deep Reinforcement Learning","date":"2018-09-13","arxiv_id":"1809.04835","repositories_listed":0,"syntology":null},{"url":null,"slug":"sim-to-real-transfer-learning-using","title":"Sim-to-Real Transfer Learning using Robustified Controllers in Robotic Tasks involving Complex Dynamics","date":"2018-09-13","arxiv_id":"1809.04720","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforced-sequence-to-set-model-for","title":"A Deep Reinforced Sequence-to-Set Model for Multi-Label Text Classification","date":"2018-09-10","arxiv_id":"1809.03118","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-behavior-generation-for-autonomous","title":"Adaptive Behavior Generation for Autonomous Driving using Deep Reinforcement Learning with Compact Semantic States","date":"2018-09-10","arxiv_id":"1809.03214","repositories_listed":0,"syntology":null}],"record_sha256":"5319f3e483c8cc92255e774bbd036bbe9473a8f4bdff524f8ebf69d9c10a915d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}