{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/129","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":129,"pages_in_order":132,"rows_per_page":100,"rows":[12801,12900],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/128","next":"/task/reinforcement-learning/papers/130","papers":[{"url":null,"slug":"is-the-bellman-residual-a-bad-proxy","title":"Is the Bellman residual a bad proxy?","date":"2016-06-24","arxiv_id":"1606.07636","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-stage-temporal-difference-learning-for","title":"Multi-Stage Temporal Difference Learning for 2048-like Games","date":"2016-06-23","arxiv_id":"1606.07374","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-preprocessing-for-tactile-data","title":"Unsupervised preprocessing for Tactile Data","date":"2016-06-23","arxiv_id":"1606.07312","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneous-control-and-human-feedback-in","title":"Simultaneous Control and Human Feedback in the Training of a Robotic Agent with Actor-Critic Reinforcement Learning","date":"2016-06-22","arxiv_id":"1606.06979","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualizing-dynamics-from-t-sne-to-semi-mdps","title":"Visualizing Dynamics: from t-SNE to SEMI-MDPs","date":"2016-06-22","arxiv_id":"1606.07112","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-hierarchical-reinforcement-learning-method","title":"A Hierarchical Reinforcement Learning Method for Persistent Time-Sensitive Tasks","date":"2016-06-20","arxiv_id":"1606.06355","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-reward-function-for-survival","title":"On Reward Function for Survival","date":"2016-06-18","arxiv_id":"1606.05767","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-discovers","title":"Deep Reinforcement Learning Discovers Internal Models","date":"2016-06-16","arxiv_id":"1606.05174","repositories_listed":0,"syntology":null},{"url":null,"slug":"successor-features-for-transfer-in","title":"Successor Features for Transfer in Reinforcement Learning","date":"2016-06-16","arxiv_id":"1606.05312","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-macro","title":"Deep Reinforcement Learning With Macro-Actions","date":"2016-06-15","arxiv_id":"1606.04615","repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-language-generation-as-planning-under","title":"Natural Language Generation as Planning under Uncertainty Using Reinforcement Learning","date":"2016-06-15","arxiv_id":"1606.04686","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategic-attentive-writer-for-learning-macro","title":"Strategic Attentive Writer for Learning Macro-Actions","date":"2016-06-15","arxiv_id":"1606.04695","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-networks-with-two-stage-training-for","title":"Policy Networks with Two-Stage Training for Dialogue Systems","date":"2016-06-10","arxiv_id":"1606.03152","repositories_listed":0,"syntology":null},{"url":null,"slug":"face-valuing-training-user-interfaces-with","title":"Face valuing: Training user interfaces with facial expressions and reinforcement learning","date":"2016-06-09","arxiv_id":"1606.02807","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuously-learning-neural-dialogue","title":"Continuously Learning Neural Dialogue Management","date":"2016-06-08","arxiv_id":"1606.02689","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapting-sampling-interval-of-sensor-networks","title":"Adapting Sampling Interval of Sensor Networks Using On-Line Reinforcement Learning","date":"2016-06-07","arxiv_id":"1606.02193","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-optimize","title":"Learning to Optimize","date":"2016-06-06","arxiv_id":"1606.01885","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-q-networks-for-accelerating-the-training","title":"Deep Q-Networks for Accelerating the Training of Deep Neural Networks","date":"2016-06-05","arxiv_id":"1606.01467","repositories_listed":0,"syntology":null},{"url":null,"slug":"difference-of-convex-functions-programming","title":"Difference of Convex Functions Programming Applied to Control with Expert Data","date":"2016-06-03","arxiv_id":"1606.01128","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-lstm-based-dialog-control","title":"End-to-end LSTM-based dialog control optimized with supervised and reinforcement learning","date":"2016-06-03","arxiv_id":"1606.01269","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-semantic","title":"Reinforcement Learning for Semantic Segmentation in Indoor Scenes","date":"2016-06-03","arxiv_id":"1606.01178","repositories_listed":0,"syntology":null},{"url":null,"slug":"death-and-suicide-in-universal-artificial","title":"Death and Suicide in Universal Artificial Intelligence","date":"2016-06-02","arxiv_id":"1606.00652","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-visual-object","title":"Reinforcement Learning for Visual Object Detection","date":"2016-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"information-theoretically-aided-reinforcement","title":"Information Theoretically Aided Reinforcement Learning for Embodied Agents","date":"2016-05-31","arxiv_id":"1605.09735","repositories_listed":0,"syntology":null},{"url":null,"slug":"kernel-mean-embedding-of-distributions-a","title":"Kernel Mean Embedding of Distributions: A Review and Beyond","date":"2016-05-31","arxiv_id":"1605.09522","repositories_listed":0,"syntology":null},{"url":null,"slug":"control-of-memory-active-perception-and","title":"Control of Memory, Active Perception, and Action in Minecraft","date":"2016-05-30","arxiv_id":"1605.09128","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-free-imitation-learning-with-policy","title":"Model-Free Imitation Learning with Policy Optimization","date":"2016-05-26","arxiv_id":"1605.08478","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-pac-rl-algorithm-for-episodic-pomdps","title":"A PAC RL Algorithm for Episodic POMDPs","date":"2016-05-25","arxiv_id":"1605.08062","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-purposeful-behaviour-in-the-absence","title":"Learning Purposeful Behaviour in the Absence of Rewards","date":"2016-05-25","arxiv_id":"1605.07700","repositories_listed":0,"syntology":null},{"url":null,"slug":"alternating-optimisation-and-quadrature-for","title":"Alternating Optimisation and Quadrature for Robust Control","date":"2016-05-24","arxiv_id":"1605.07496","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-memory-networks","title":"Hierarchical Memory Networks","date":"2016-05-24","arxiv_id":"1605.07427","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-line-active-reward-learning-for-policy","title":"On-line Active Reward Learning for Policy Optimisation in Spoken Dialogue Systems","date":"2016-05-24","arxiv_id":"1605.07669","repositories_listed":0,"syntology":null},{"url":null,"slug":"localizing-by-describing-attribute-guided","title":"Localizing by Describing: Attribute-Guided Attention Localization for Fine-Grained Recognition","date":"2016-05-20","arxiv_id":"1605.06217","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-frame-skip-deep-q-network","title":"Dynamic Frame skip Deep Q Network","date":"2016-05-17","arxiv_id":"1605.05365","repositories_listed":0,"syntology":null},{"url":null,"slug":"option-discovery-in-hierarchical","title":"Option Discovery in Hierarchical Reinforcement Learning using Spatio-Temporal Clustering","date":"2016-05-17","arxiv_id":"1605.05359","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-system-to-encourage","title":"A Reinforcement Learning System to Encourage Physical Activity in Diabetes Patients","date":"2016-05-13","arxiv_id":"1605.04070","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-human-interpretable-dialog","title":"Optimizing human-interpretable dialog management policy using Genetic Algorithm","date":"2016-05-12","arxiv_id":"1605.03915","repositories_listed":0,"syntology":null},{"url":null,"slug":"avoiding-wireheading-with-value-reinforcement","title":"Avoiding Wireheading with Value Reinforcement Learning","date":"2016-05-10","arxiv_id":"1605.03143","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-modification-of-policy-and-utility","title":"Self-Modification of Policy and Utility Function in Rational Agents","date":"2016-05-10","arxiv_id":"1605.03142","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-self-taught-artificial-agent-for-multi","title":"A Self-Taught Artificial Agent for Multi-Physics Computational Model Personalization","date":"2016-05-01","arxiv_id":"1605.00303","repositories_listed":0,"syntology":null},{"url":null,"slug":"classifying-options-for-deep-reinforcement","title":"Classifying Options for Deep Reinforcement Learning","date":"2016-04-27","arxiv_id":"1604.08153","repositories_listed":0,"syntology":null},{"url":null,"slug":"tournament-selection-in-zeroth-level","title":"Tournament selection in zeroth-level classifier systems based on average reward reinforcement learning","date":"2016-04-26","arxiv_id":"1604.07704","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-hierarchical-approach-to-lifelong","title":"A Deep Hierarchical Approach to Lifelong Learning in Minecraft","date":"2016-04-25","arxiv_id":"1604.07255","repositories_listed":0,"syntology":null},{"url":null,"slug":"neurohex-a-deep-q-learning-hex-agent","title":"Neurohex: A Deep Q-learning Hex Agent","date":"2016-04-24","arxiv_id":"1604.07097","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-reinforcement-learning-to-validate","title":"Using Reinforcement Learning to Validate Empirical Game-Theoretic Analysis: A Continuous Double Auction Study","date":"2016-04-22","arxiv_id":"1604.06710","repositories_listed":0,"syntology":null},{"url":null,"slug":"closed-loop-interactions-between-spiking","title":"Closed loop interactions between spiking neural network and robotic simulators based on MUSIC and ROS","date":"2016-04-16","arxiv_id":"1604.04764","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-generate-genotypes-with-neural","title":"Learning to Generate Genotypes with Neural Networks","date":"2016-04-14","arxiv_id":"1604.04153","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-with-1","title":"Inverse Reinforcement Learning with Simultaneous Estimation of Rewards and Dynamics","date":"2016-04-13","arxiv_id":"1604.03912","repositories_listed":0,"syntology":null},{"url":null,"slug":"theoretically-grounded-policy-advice-from","title":"Theoretically-Grounded Policy Advice from Multiple Teachers in Reinforcement Learning Settings with Applications to Negative Transfer","date":"2016-04-13","arxiv_id":"1604.03986","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-statistical-learning-strategy-for-closed","title":"A statistical learning strategy for closed-loop control of fluid flows","date":"2016-04-11","arxiv_id":"1604.03392","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-based-local-search-for","title":"Reinforcement learning based local search for grouping problems: A case study on graph coloring","date":"2016-04-01","arxiv_id":"1604.00377","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-recommendation-to-users-that-react","title":"Optimal Recommendation to Users that React: Online Learning for a Class of POMDPs","date":"2016-03-30","arxiv_id":"1603.09233","repositories_listed":0,"syntology":null},{"url":null,"slug":"algorithms-for-batch-hierarchical","title":"Algorithms for Batch Hierarchical Reinforcement Learning","date":"2016-03-29","arxiv_id":"1603.08869","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-visual-explanations","title":"Generating Visual Explanations","date":"2016-03-28","arxiv_id":"1603.08507","repositories_listed":0,"syntology":null},{"url":null,"slug":"negative-learning-rates-and-p-learning","title":"Negative Learning Rates and P-Learning","date":"2016-03-27","arxiv_id":"1603.08253","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-parameter-selection-in-evolutionary","title":"Adaptive Parameter Selection in Evolutionary Algorithms by Reinforcement Learning with Dynamic Discretization of Parameter Range","date":"2016-03-22","arxiv_id":"1603.06788","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-convolutional-attention-networks-for","title":"Fully Convolutional Attention Networks for Fine-Grained Recognition","date":"2016-03-22","arxiv_id":"1603.06765","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-dexterous-manipulation-for-a-soft","title":"Learning Dexterous Manipulation for a Soft Robotic Hand from Human Demonstration","date":"2016-03-21","arxiv_id":"1603.06348","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-selection-as-a-multiagent","title":"Feature Selection as a Multiagent Coordination Problem","date":"2016-03-16","arxiv_id":"1603.05152","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-signaling-game-approach-to-databases","title":"A Signaling Game Approach to Databases Querying and Interaction","date":"2016-03-13","arxiv_id":"1603.04068","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-linearly-solvable-markov","title":"Hierarchical Linearly-Solvable Markov Decision Problems","date":"2016-03-10","arxiv_id":"1603.03267","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-autoencoders-for-semi-supervised","title":"Variational Autoencoders for Semi-supervised Text Classification","date":"2016-03-08","arxiv_id":"1603.02514","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentially-private-policy-evaluation","title":"Differentially Private Policy Evaluation","date":"2016-03-07","arxiv_id":"1603.02010","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-shared-representations-in-multi-task","title":"Learning Shared Representations in Multi-task Reinforcement Learning","date":"2016-03-07","arxiv_id":"1603.02041","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-decision-making-in-electricity","title":"Hierarchical Decision Making In Electricity Grid Management","date":"2016-03-06","arxiv_id":"1603.01840","repositories_listed":0,"syntology":null},{"url":null,"slug":"object-manipulation-learning-by-imitation","title":"Object Manipulation Learning by Imitation","date":"2016-03-03","arxiv_id":"1603.00964","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-information-source-optimization","title":"Multi-Information Source Optimization","date":"2016-03-01","arxiv_id":"1603.00389","repositories_listed":0,"syntology":null},{"url":null,"slug":"easy-monotonic-policy-iteration","title":"Easy Monotonic Policy Iteration","date":"2016-02-29","arxiv_id":"1602.09118","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-within-projective-simulation","title":"Meta-learning within Projective Simulation","date":"2016-02-25","arxiv_id":"1602.08017","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-of-pomdps-using","title":"Reinforcement Learning of POMDPs using Spectral Methods","date":"2016-02-25","arxiv_id":"1602.07764","repositories_listed":0,"syntology":null},{"url":null,"slug":"thompson-sampling-is-asymptotically-optimal","title":"Thompson Sampling is Asymptotically Optimal in General Environments","date":"2016-02-25","arxiv_id":"1602.07905","repositories_listed":0,"syntology":null},{"url":"/paper/learning-values-across-many-orders-of","slug":"learning-values-across-many-orders-of","title":"Learning values across many orders of magnitude","date":"2016-02-24","arxiv_id":"1602.07714","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-error-bounds-for-model-based","title":"Policy Error Bounds for Model-Based Reinforcement Learning with Factored Linear Models","date":"2016-02-19","arxiv_id":"1602.06346","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-in-swarm","title":"Inverse Reinforcement Learning in Swarm Systems","date":"2016-02-17","arxiv_id":"1602.05450","repositories_listed":0,"syntology":null},{"url":null,"slug":"pomdp-lite-for-robust-robot-planning-under","title":"POMDP-lite for Robust Robot Planning under Uncertainty","date":"2016-02-16","arxiv_id":"1602.04875","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-approach-for-real-time","title":"Reinforcement Learning approach for Real Time Strategy Games Battle city and S3","date":"2016-02-16","arxiv_id":"1602.04936","repositories_listed":0,"syntology":null},{"url":null,"slug":"training-of-spiking-neural-networks-based-on","title":"Training of spiking neural networks based on information theoretic costs","date":"2016-02-15","arxiv_id":"1602.04742","repositories_listed":0,"syntology":null},{"url":null,"slug":"using-deep-q-learning-to-control-optimization","title":"Using Deep Q-Learning to Control Optimization Hyperparameters","date":"2016-02-12","arxiv_id":"1602.04062","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-efficient-reinforcement-learning-in-1","title":"Data-Efficient Reinforcement Learning in Continuous-State POMDPs","date":"2016-02-08","arxiv_id":"1602.02523","repositories_listed":0,"syntology":null},{"url":null,"slug":"graying-the-black-box-understanding-dqns","title":"Graying the black box: Understanding DQNs","date":"2016-02-08","arxiv_id":"1602.02658","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-communicate-to-solve-riddles-with","title":"Learning to Communicate to Solve Riddles with Deep Distributed Recurrent Q-Networks","date":"2016-02-08","arxiv_id":"1602.02672","repositories_listed":0,"syntology":null},{"url":null,"slug":"pac-reinforcement-learning-with-rich","title":"PAC Reinforcement Learning with Rich Observations","date":"2016-02-08","arxiv_id":"1602.02722","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-information-acquisition","title":"Active Information Acquisition","date":"2016-02-05","arxiv_id":"1602.02181","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantum-machine-learning-with-glow-for","title":"Quantum machine learning with glow for episodic tasks and decision games","date":"2016-01-27","arxiv_id":"1601.07358","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-resolving-unidentifiability-in","title":"Towards Resolving Unidentifiability in Inverse Reinforcement Learning","date":"2016-01-25","arxiv_id":"1601.06569","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-via-deep","title":"Inverse Reinforcement Learning via Deep Gaussian Process","date":"2015-12-26","arxiv_id":"1512.08065","repositories_listed":0,"syntology":null},{"url":null,"slug":"information-theoretic-bounded-rationality","title":"Information-Theoretic Bounded Rationality","date":"2015-12-21","arxiv_id":"1512.06789","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-comparison-of-neural","title":"An Empirical Comparison of Neural Architectures for Reinforcement Learning in Partially Observable Environments","date":"2015-12-17","arxiv_id":"1512.05509","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-active-object-recognition-by-joint-label","title":"Deep Active Object Recognition by Joint Label and Action Prediction","date":"2015-12-17","arxiv_id":"1512.05484","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-discount-deep-reinforcement-learning","title":"How to Discount Deep Reinforcement Learning: Towards New Dynamic Strategies","date":"2015-12-07","arxiv_id":"1512.02011","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-constrained-reinforcement-learning-with","title":"Risk-Constrained Reinforcement Learning with Percentile Risk Criteria","date":"2015-12-05","arxiv_id":"1512.01629","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-networks-for-binary-vector-actions","title":"Q-Networks for Binary Vector Actions","date":"2015-12-04","arxiv_id":"1512.01332","repositories_listed":0,"syntology":null},{"url":null,"slug":"reuse-of-neural-modules-for-general-video","title":"Reuse of Neural Modules for General Video Game Playing","date":"2015-12-04","arxiv_id":"1512.01537","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-with-attention","title":"Deep Reinforcement Learning with Attention for Slate Markov Decision Processes with High-Dimensional States and Actions","date":"2015-12-03","arxiv_id":"1512.01124","repositories_listed":0,"syntology":null},{"url":null,"slug":"inverse-reinforcement-learning-with-locally","title":"Inverse Reinforcement Learning with Locally Consistent Reward Functions","date":"2015-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/learning-to-track-online-multi-object","slug":"learning-to-track-online-multi-object","title":"Learning to Track: Online Multi-Object Tracking by Decision Making","date":"2015-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-class-multi-annotator-active-learning","title":"Multi-Class Multi-Annotator Active Learning With Robust Gaussian Process for Visual Recognition","date":"2015-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-learning-to-think-algorithmic-information","title":"On Learning to Think: Algorithmic Information Theory for Novel Combinations of Reinforcement Learning Controllers and Recurrent Neural World Models","date":"2015-11-30","arxiv_id":"1511.09249","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-applied-to-an-electric","title":"Reinforcement Learning Applied to an Electric Water Heater: From Theory to Practice","date":"2015-11-29","arxiv_id":"1512.00408","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-search-rescue-via-online-multi-task","title":"Robotic Search & Rescue via Online Multi-task Reinforcement Learning","date":"2015-11-29","arxiv_id":"1511.08967","repositories_listed":0,"syntology":null}],"record_sha256":"2e2d9393756227e5e78a4ff38ee6ca5ba3c6c257e193bf585e05381d53666f79","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}