{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/17","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":17,"pages_in_order":59,"rows_per_page":100,"rows":[1601,1700],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/16","next":"/task/deep-reinforcement-learning/papers/18","papers":[{"url":"/paper/sfv-reinforcement-learning-of-physical-skills","slug":"sfv-reinforcement-learning-of-physical-skills","title":"SFV: Reinforcement Learning of Physical Skills from Videos","date":"2018-10-08","arxiv_id":"1810.03599","repositories_listed":1,"syntology":null},{"url":"/paper/where-did-my-optimum-go-an-empirical-analysis","slug":"where-did-my-optimum-go-an-empirical-analysis","title":"Where Did My Optimum Go?: An Empirical Analysis of Gradient Descent Optimization in Policy Gradient Methods","date":"2018-10-05","arxiv_id":"1810.02525","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/where-did-my-optimum-go-an-empirical-analysis#ran","syntology_url":"https://syntology.ai/paper/1810.02525","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.02525"}},"official":{"repos":["facebookresearch/WhereDidMyOptimumGo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/image-based-guidance-of-autonomous-aircraft","slug":"image-based-guidance-of-autonomous-aircraft","title":"Image-based Guidance of Autonomous Aircraft for Wildfire Surveillance and Prediction","date":"2018-10-04","arxiv_id":"1810.02455","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-perturbed-rewards","slug":"reinforcement-learning-with-perturbed-rewards","title":"Reinforcement Learning with Perturbed Rewards","date":"2018-10-02","arxiv_id":"1810.01032","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-perturbed-rewards#ran","syntology_url":"https://syntology.ai/paper/1810.01032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.01032"}},"official":{"repos":["wangjksjtu/rl-perturbed-reward"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/interactive-learning-with-corrective-feedback","slug":"interactive-learning-with-corrective-feedback","title":"Interactive Learning with Corrective Feedback for Policies based on Deep Neural Networks","date":"2018-09-30","arxiv_id":"1810.00466","repositories_listed":1,"syntology":null},{"url":"/paper/using-state-predictions-for-value","slug":"using-state-predictions-for-value","title":"Using State Predictions for Value Regularization in Curiosity Driven Deep Reinforcement Learning","date":"2018-09-30","arxiv_id":"1810.00361","repositories_listed":1,"syntology":null},{"url":"/paper/generalization-and-regularization-in-dqn","slug":"generalization-and-regularization-in-dqn","title":"Generalization and Regularization in DQN","date":"2018-09-29","arxiv_id":"1810.00123","repositories_listed":1,"syntology":null},{"url":"/paper/propagation-networks-for-model-based-control","slug":"propagation-networks-for-model-based-control","title":"Propagation Networks for Model-Based Control Under Partial Observation","date":"2018-09-28","arxiv_id":"1809.11169","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-contact-forces-for-learning-to","slug":"leveraging-contact-forces-for-learning-to","title":"Leveraging Contact Forces for Learning to Grasp","date":"2018-09-19","arxiv_id":"1809.07004","repositories_listed":1,"syntology":null},{"url":"/paper/generalizing-across-multi-objective-reward","slug":"generalizing-across-multi-objective-reward","title":"Generalizing Across Multi-Objective Reward Functions in Deep Reinforcement Learning","date":"2018-09-17","arxiv_id":"1809.06364","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/generalizing-across-multi-objective-reward#ran","syntology_url":"https://syntology.ai/paper/1809.06364","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.06364"}},"official":null}},{"url":"/paper/muscle-excitation-estimation-in-biomechanical","slug":"muscle-excitation-estimation-in-biomechanical","title":"Muscle Excitation Estimation in Biomechanical Simulation Using NAF Reinforcement Learning","date":"2018-09-17","arxiv_id":"1809.06121","repositories_listed":1,"syntology":null},{"url":"/paper/transparency-and-explanation-in-deep","slug":"transparency-and-explanation-in-deep","title":"Transparency and Explanation in Deep Reinforcement Learning Neural Networks","date":"2018-09-17","arxiv_id":"1809.06061","repositories_listed":1,"syntology":null},{"url":"/paper/deterministic-implementations-for","slug":"deterministic-implementations-for","title":"Deterministic Implementations for Reproducibility in Deep Reinforcement Learning","date":"2018-09-15","arxiv_id":"1809.05676","repositories_listed":1,"syntology":null},{"url":"/paper/towards-better-interpretability-in-deep-q","slug":"towards-better-interpretability-in-deep-q","title":"Towards Better Interpretability in Deep Q-Networks","date":"2018-09-15","arxiv_id":"1809.05630","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-better-interpretability-in-deep-q#ran","syntology_url":"https://syntology.ai/paper/1809.05630","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.05630"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-for-event","slug":"deep-reinforcement-learning-for-event","title":"Deep Reinforcement Learning for Event-Triggered Control","date":"2018-09-13","arxiv_id":"1809.05152","repositories_listed":1,"syntology":null},{"url":"/paper/improving-optimization-bounds-using-machine","slug":"improving-optimization-bounds-using-machine","title":"Improving Optimization Bounds using Machine Learning: Decision Diagrams meet Deep Reinforcement Learning","date":"2018-09-10","arxiv_id":"1809.03359","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/improving-optimization-bounds-using-machine#ran","syntology_url":"https://syntology.ai/paper/1809.03359","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.03359"}},"official":{"repos":["qcappart/learning-DD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/keep-it-stupid-simple","slug":"keep-it-stupid-simple","title":"Combining imagination and heuristics to learn strategies that generalize","date":"2018-09-10","arxiv_id":"1809.03406","repositories_listed":1,"syntology":null},{"url":"/paper/archer-aggressive-rewards-to-counter-bias-in","slug":"archer-aggressive-rewards-to-counter-bias-in","title":"ARCHER: Aggressive Rewards to Counter bias in Hindsight Experience Replay","date":"2018-09-06","arxiv_id":"1809.02070","repositories_listed":1,"syntology":null},{"url":"/paper/challenges-of-context-and-time-in","slug":"challenges-of-context-and-time-in","title":"Challenges of Context and Time in Reinforcement Learning: Introducing Space Fortress as a Benchmark","date":"2018-09-06","arxiv_id":"1809.02206","repositories_listed":1,"syntology":null},{"url":"/paper/visual-transfer-between-atari-games-using","slug":"visual-transfer-between-atari-games-using","title":"Visual Transfer between Atari Games using Competitive Reinforcement Learning","date":"2018-09-02","arxiv_id":"1809.00397","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/visual-transfer-between-atari-games-using#ran","syntology_url":"https://syntology.ai/paper/1809.00397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1809.00397"}},"official":{"repos":["sowmya-mp/rl_a3c_pytorch"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/exit-oos-towards-learning-from-planning-in","slug":"exit-oos-towards-learning-from-planning-in","title":"ExIt-OOS: Towards Learning from Planning in Imperfect Information Games","date":"2018-08-30","arxiv_id":"1808.10120","repositories_listed":1,"syntology":null},{"url":"/paper/context-aware-visual-policy-network-for","slug":"context-aware-visual-policy-network-for","title":"Context-Aware Visual Policy Network for Sequence-Level Image Captioning","date":"2018-08-16","arxiv_id":"1808.05864","repositories_listed":1,"syntology":null},{"url":"/paper/deep-rts-a-game-environment-for-deep","slug":"deep-rts-a-game-environment-for-deep","title":"Deep RTS: A Game Environment for Deep Reinforcement Learning in Real-Time Strategy Games","date":"2018-08-15","arxiv_id":"1808.05032","repositories_listed":1,"syntology":null},{"url":"/paper/an-efficient-deep-reinforcement-learning","slug":"an-efficient-deep-reinforcement-learning","title":"An Efficient Deep Reinforcement Learning Model for Urban Traffic Control","date":"2018-08-06","arxiv_id":"1808.01876","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-deep-reinforcement-learning-for-1","slug":"multi-agent-deep-reinforcement-learning-for-1","title":"Multi-Agent Deep Reinforcement Learning for Dynamic Power Allocation in Wireless Networks","date":"2018-08-01","arxiv_id":"1808.00490","repositories_listed":1,"syntology":null},{"url":"/paper/learning-heuristics-for-automated-reasoning","slug":"learning-heuristics-for-automated-reasoning","title":"Learning Heuristics for Quantified Boolean Formulas through Deep Reinforcement Learning","date":"2018-07-20","arxiv_id":"1807.08058","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-swarm-systems","slug":"deep-reinforcement-learning-for-swarm-systems","title":"Deep Reinforcement Learning for Swarm Systems","date":"2018-07-17","arxiv_id":"1807.06613","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-listen-read-and-follow-score","slug":"learning-to-listen-read-and-follow-score","title":"Learning to Listen, Read, and Follow: Score Following as a Reinforcement Learning Game","date":"2018-07-17","arxiv_id":"1807.06391","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/learning-to-listen-read-and-follow-score#ran","syntology_url":"https://syntology.ai/paper/1807.06391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1807.06391"}},"official":{"repos":["CPJKU/score_following_game"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/financial-trading-as-a-game-a-deep","slug":"financial-trading-as-a-game-a-deep","title":"Financial Trading as a Game: A Deep Reinforcement Learning Approach","date":"2018-07-08","arxiv_id":"1807.02787","repositories_listed":1,"syntology":null},{"url":"/paper/learning-goal-oriented-visual-dialog-via","slug":"learning-goal-oriented-visual-dialog-via","title":"Learning Goal-Oriented Visual Dialog via Tempered Policy Gradient","date":"2018-07-02","arxiv_id":"1807.00737","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-in-continuous","slug":"deep-reinforcement-learning-in-continuous","title":"Deep Reinforcement Learning in Continuous Action Spaces: a Case Study in the Game of Simulated Curling","date":"2018-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/illuminating-generalization-in-deep","slug":"illuminating-generalization-in-deep","title":"Illuminating Generalization in Deep Reinforcement Learning through Procedural Level Generation","date":"2018-06-28","arxiv_id":"1806.10729","repositories_listed":1,"syntology":null},{"url":"/paper/qt-opt-scalable-deep-reinforcement-learning","slug":"qt-opt-scalable-deep-reinforcement-learning","title":"QT-Opt: Scalable Deep Reinforcement Learning for Vision-Based Robotic Manipulation","date":"2018-06-27","arxiv_id":"1806.10293","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-surgical","slug":"deep-reinforcement-learning-for-surgical","title":"Deep Reinforcement Learning for Surgical Gesture Segmentation and Classification","date":"2018-06-21","arxiv_id":"1806.08089","repositories_listed":1,"syntology":null},{"url":"/paper/how-many-random-seeds-statistical-power","slug":"how-many-random-seeds-statistical-power","title":"How Many Random Seeds? Statistical Power Analysis in Deep Reinforcement Learning Experiments","date":"2018-06-21","arxiv_id":"1806.08295","repositories_listed":1,"syntology":null},{"url":"/paper/sim-to-real-reinforcement-learning-for","slug":"sim-to-real-reinforcement-learning-for","title":"Sim-to-Real Reinforcement Learning for Deformable Object Manipulation","date":"2018-06-20","arxiv_id":"1806.07851","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sim-to-real-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/1806.07851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.07851"}},"official":{"repos":["JanMatas/Rainbow_ddpg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/automated-image-data-preprocessing-with-deep","slug":"automated-image-data-preprocessing-with-deep","title":"Automated Image Data Preprocessing with Deep Reinforcement Learning","date":"2018-06-15","arxiv_id":"1806.05886","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-chinese-zero","slug":"deep-reinforcement-learning-for-chinese-zero","title":"Deep Reinforcement Learning for Chinese Zero pronoun Resolution","date":"2018-06-10","arxiv_id":"1806.03711","repositories_listed":1,"syntology":null},{"url":"/paper/conversational-recommender-system","slug":"conversational-recommender-system","title":"Conversational Recommender System","date":"2018-06-08","arxiv_id":"1806.03277","repositories_listed":1,"syntology":null},{"url":"/paper/randomized-prior-functions-for-deep","slug":"randomized-prior-functions-for-deep","title":"Randomized Prior Functions for Deep Reinforcement Learning","date":"2018-06-08","arxiv_id":"1806.03335","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/randomized-prior-functions-for-deep#ran","syntology_url":"https://syntology.ai/paper/1806.03335","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1806.03335"}},"official":null}},{"url":"/paper/learning-to-understand-goal-specifications-by","slug":"learning-to-understand-goal-specifications-by","title":"Learning to Understand Goal Specifications by Modelling Reward","date":"2018-06-05","arxiv_id":"1806.01946","repositories_listed":1,"syntology":null},{"url":"/paper/playing-atari-with-six-neurons","slug":"playing-atari-with-six-neurons","title":"Playing Atari with Six Neurons","date":"2018-06-04","arxiv_id":"1806.01363","repositories_listed":1,"syntology":null},{"url":"/paper/td-or-not-td-analyzing-the-role-of-temporal","slug":"td-or-not-td-analyzing-the-role-of-temporal","title":"TD or not TD: Analyzing the Role of Temporal Differencing in Deep Reinforcement Learning","date":"2018-06-04","arxiv_id":"1806.01175","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-of-region","slug":"deep-reinforcement-learning-of-region","title":"Deep Reinforcement Learning of Region Proposal Networks for Object Detection","date":"2018-06-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-deep-reinforcement-learning-2","slug":"sample-efficient-deep-reinforcement-learning-2","title":"Sample-Efficient Deep Reinforcement Learning via Episodic Backward Update","date":"2018-05-31","arxiv_id":"1805.12375","repositories_listed":1,"syntology":null},{"url":"/paper/playing-hard-exploration-games-by-watching","slug":"playing-hard-exploration-games-by-watching","title":"Playing hard exploration games by watching YouTube","date":"2018-05-29","arxiv_id":"1805.11592","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/playing-hard-exploration-games-by-watching#ran","syntology_url":"https://syntology.ai/paper/1805.11592","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.11592"}},"official":null}},{"url":"/paper/supervised-policy-update-for-deep","slug":"supervised-policy-update-for-deep","title":"Supervised Policy Update for Deep Reinforcement Learning","date":"2018-05-29","arxiv_id":"1805.11706","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-of-marked","slug":"deep-reinforcement-learning-of-marked","title":"Deep Reinforcement Learning of Marked Temporal Point Processes","date":"2018-05-23","arxiv_id":"1805.09360","repositories_listed":1,"syntology":null},{"url":"/paper/guided-feature-transformation-gft-a-neural","slug":"guided-feature-transformation-gft-a-neural","title":"Guided Feature Transformation (GFT): A Neural Language Grounding Module for Embodied Agents","date":"2018-05-22","arxiv_id":"1805.08329","repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-video-object-segmentation-for","slug":"unsupervised-video-object-segmentation-for","title":"Unsupervised Video Object Segmentation for Deep Reinforcement Learning","date":"2018-05-20","arxiv_id":"1805.07780","repositories_listed":1,"syntology":null},{"url":"/paper/do-deep-reinforcement-learning-agents-model","slug":"do-deep-reinforcement-learning-agents-model","title":"Do deep reinforcement learning agents model intentions?","date":"2018-05-15","arxiv_id":"1805.06020","repositories_listed":1,"syntology":null},{"url":"/paper/general-solutions-for-nonlinear-differential","slug":"general-solutions-for-nonlinear-differential","title":"General solutions for nonlinear differential equations: a rule-based self-learning approach using deep reinforcement learning","date":"2018-05-13","arxiv_id":"1805.07297","repositories_listed":1,"syntology":null},{"url":"/paper/reward-estimation-for-variance-reduction-in","slug":"reward-estimation-for-variance-reduction-in","title":"Reward Estimation for Variance Reduction in Deep Reinforcement Learning","date":"2018-05-09","arxiv_id":"1805.03359","repositories_listed":1,"syntology":null},{"url":"/paper/ranking-for-relevance-and-display-preferences","slug":"ranking-for-relevance-and-display-preferences","title":"Ranking for Relevance and Display Preferences in Complex Presentation Layouts","date":"2018-05-07","arxiv_id":"1805.02404","repositories_listed":1,"syntology":null},{"url":"/paper/towards-symbolic-reinforcement-learning-with","slug":"towards-symbolic-reinforcement-learning-with","title":"Towards Symbolic Reinforcement Learning with Common Sense","date":"2018-04-23","arxiv_id":"1804.08597","repositories_listed":1,"syntology":null},{"url":"/paper/a-study-on-overfitting-in-deep-reinforcement","slug":"a-study-on-overfitting-in-deep-reinforcement","title":"A Study on Overfitting in Deep Reinforcement Learning","date":"2018-04-18","arxiv_id":"1804.06893","repositories_listed":1,"syntology":null},{"url":"/paper/look-before-you-leap-bridging-model-free-and","slug":"look-before-you-leap-bridging-model-free-and","title":"Look Before You Leap: Bridging Model-Free and Model-Based Reinforcement Learning for Planned-Ahead Vision-and-Language Navigation","date":"2018-03-21","arxiv_id":"1803.07729","repositories_listed":1,"syntology":null},{"url":"/paper/composable-deep-reinforcement-learning-for","slug":"composable-deep-reinforcement-learning-for","title":"Composable Deep Reinforcement Learning for Robotic Manipulation","date":"2018-03-19","arxiv_id":"1803.06773","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-play-general-video-games-via-an","slug":"learning-to-play-general-video-games-via-an","title":"Learning to Play General Video-Games via an Object Embedding Network","date":"2018-03-14","arxiv_id":"1803.05262","repositories_listed":1,"syntology":null},{"url":"/paper/the-advantage-of-doubling-a-deep","slug":"the-advantage-of-doubling-a-deep","title":"The Advantage of Doubling: A Deep Reinforcement Learning Approach to Studying the Double Team in the NBA","date":"2018-03-08","arxiv_id":"1803.02940","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-vision-based","slug":"deep-reinforcement-learning-for-vision-based","title":"Deep Reinforcement Learning for Vision-Based Robotic Grasping: A Simulated Comparative Evaluation of Off-Policy Methods","date":"2018-02-28","arxiv_id":"1802.10264","repositories_listed":1,"syntology":null},{"url":"/paper/selective-experience-replay-for-lifelong","slug":"selective-experience-replay-for-lifelong","title":"Selective Experience Replay for Lifelong Learning","date":"2018-02-28","arxiv_id":"1802.10269","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-and-imitation-learning-for","slug":"reinforcement-and-imitation-learning-for","title":"Reinforcement and Imitation Learning for Diverse Visuomotor Skills","date":"2018-02-26","arxiv_id":"1802.09564","repositories_listed":1,"syntology":null},{"url":"/paper/structured-control-nets-for-deep","slug":"structured-control-nets-for-deep","title":"Structured Control Nets for Deep Reinforcement Learning","date":"2018-02-22","arxiv_id":"1802.08311","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-large-scale-fleet-management-via","slug":"efficient-large-scale-fleet-management-via","title":"Efficient Collaborative Multi-Agent Deep Reinforcement Learning for Large-Scale Fleet Management","date":"2018-02-18","arxiv_id":"1802.06444","repositories_listed":1,"syntology":null},{"url":"/paper/from-gameplay-to-symbolic-reasoning-learning","slug":"from-gameplay-to-symbolic-reasoning-learning","title":"From Gameplay to Symbolic Reasoning: Learning SAT Solver Heuristics in the Style of Alpha(Go) Zero","date":"2018-02-14","arxiv_id":"1802.05340","repositories_listed":1,"syntology":null},{"url":"/paper/gep-pg-decoupling-exploration-and","slug":"gep-pg-decoupling-exploration-and","title":"GEP-PG: Decoupling Exploration and Exploitation in Deep Reinforcement Learning Algorithms","date":"2018-02-14","arxiv_id":"1802.05054","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gep-pg-decoupling-exploration-and#ran","syntology_url":"https://syntology.ai/paper/1802.05054","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1802.05054"}},"official":{"repos":["flowersteam/geppg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/efficient-model-based-deep-reinforcement","slug":"efficient-model-based-deep-reinforcement","title":"Efficient Model-Based Deep Reinforcement Learning with Variational State Tabulation","date":"2018-02-12","arxiv_id":"1802.04325","repositories_listed":1,"syntology":null},{"url":"/paper/a-critical-investigation-of-deep","slug":"a-critical-investigation-of-deep","title":"A Critical Investigation of Deep Reinforcement Learning for Navigation","date":"2018-02-07","arxiv_id":"1802.02274","repositories_listed":1,"syntology":null},{"url":"/paper/shared-autonomy-via-deep-reinforcement","slug":"shared-autonomy-via-deep-reinforcement","title":"Shared Autonomy via Deep Reinforcement Learning","date":"2018-02-06","arxiv_id":"1802.01744","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-programming","slug":"deep-reinforcement-learning-for-programming","title":"Deep Reinforcement Learning for Programming Language Correction","date":"2018-01-31","arxiv_id":"1801.10467","repositories_listed":1,"syntology":null},{"url":"/paper/psychlab-a-psychology-laboratory-for-deep","slug":"psychlab-a-psychology-laboratory-for-deep","title":"Psychlab: A Psychology Laboratory for Deep Reinforcement Learning Agents","date":"2018-01-24","arxiv_id":"1801.08116","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-deep-reinforcement-learning-learn","slug":"distributed-deep-reinforcement-learning-learn","title":"Distributed Deep Reinforcement Learning: Learn how to play Atari games in 21 minutes","date":"2018-01-09","arxiv_id":"1801.02852","repositories_listed":1,"syntology":null},{"url":"/paper/parametrized-deep-q-networks-learning-playing","slug":"parametrized-deep-q-networks-learning-playing","title":"PARAMETRIZED DEEP Q-NETWORKS LEARNING: PLAYING ONLINE BATTLE ARENA WITH DISCRETE-CONTINUOUS HYBRID ACTION SPACE","date":"2018-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/federated-control-with-hierarchical-multi","slug":"federated-control-with-hierarchical-multi","title":"Federated Control with Hierarchical Multi-Agent Deep Reinforcement Learning","date":"2017-12-22","arxiv_id":"1712.08266","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-de-novo-drug","slug":"deep-reinforcement-learning-for-de-novo-drug","title":"Deep Reinforcement Learning for De-Novo Drug Design","date":"2017-11-29","arxiv_id":"1711.10907","repositories_listed":1,"syntology":null},{"url":"/paper/divide-and-conquer-reinforcement-learning","slug":"divide-and-conquer-reinforcement-learning","title":"Divide-and-Conquer Reinforcement Learning","date":"2017-11-27","arxiv_id":"1711.09874","repositories_listed":1,"syntology":null},{"url":"/paper/classification-with-costly-features-using","slug":"classification-with-costly-features-using","title":"Classification with Costly Features using Deep Reinforcement Learning","date":"2017-11-20","arxiv_id":"1711.07364","repositories_listed":1,"syntology":null},{"url":"/paper/implementing-the-deep-q-network","slug":"implementing-the-deep-q-network","title":"Implementing the Deep Q-Network","date":"2017-11-20","arxiv_id":"1711.07478","repositories_listed":1,"syntology":null},{"url":"/paper/teaching-a-machine-to-read-maps-with-deep","slug":"teaching-a-machine-to-read-maps-with-deep","title":"Teaching a Machine to Read Maps with Deep Reinforcement Learning","date":"2017-11-20","arxiv_id":"1711.07479","repositories_listed":1,"syntology":null},{"url":"/paper/leave-no-trace-learning-to-reset-for-safe-and","slug":"leave-no-trace-learning-to-reset-for-safe-and","title":"Leave no Trace: Learning to Reset for Safe and Autonomous Reinforcement Learning","date":"2017-11-18","arxiv_id":"1711.06782","repositories_listed":1,"syntology":null},{"url":"/paper/towards-the-use-of-deep-reinforcement","slug":"towards-the-use-of-deep-reinforcement","title":"Towards the Use of Deep Reinforcement Learning with Global Policy For Query-based Extractive Summarisation","date":"2017-11-10","arxiv_id":"1711.03859","repositories_listed":1,"syntology":null},{"url":"/paper/can-deep-reinforcement-learning-solve-erdos","slug":"can-deep-reinforcement-learning-solve-erdos","title":"Can Deep Reinforcement Learning Solve Erdos-Selfridge-Spencer Games?","date":"2017-11-07","arxiv_id":"1711.02301","repositories_listed":1,"syntology":null},{"url":"/paper/a-unified-game-theoretic-approach-to","slug":"a-unified-game-theoretic-approach-to","title":"A Unified Game-Theoretic Approach to Multiagent Reinforcement Learning","date":"2017-11-02","arxiv_id":"1711.00832","repositories_listed":1,"syntology":null},{"url":"/paper/regret-minimization-for-partially-observable","slug":"regret-minimization-for-partially-observable","title":"Regret Minimization for Partially Observable Deep Reinforcement Learning","date":"2017-10-31","arxiv_id":"1710.11424","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regret-minimization-for-partially-observable#ran","syntology_url":"https://syntology.ai/paper/1710.11424","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11424"}},"official":null}},{"url":"/paper/treeqn-and-atreec-differentiable-tree","slug":"treeqn-and-atreec-differentiable-tree","title":"TreeQN and ATreeC: Differentiable Tree-Structured Models for Deep Reinforcement Learning","date":"2017-10-31","arxiv_id":"1710.11417","repositories_listed":1,"syntology":{"n":12,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/treeqn-and-atreec-differentiable-tree#ran","syntology_url":"https://syntology.ai/paper/1710.11417","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11417"}},"official":{"repos":["oxwhirl/treeqn"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/eigenoption-discovery-through-the-deep","slug":"eigenoption-discovery-through-the-deep","title":"Eigenoption Discovery through the Deep Successor Representation","date":"2017-10-30","arxiv_id":"1710.11089","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/eigenoption-discovery-through-the-deep#ran","syntology_url":"https://syntology.ai/paper/1710.11089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1710.11089"}},"official":null}},{"url":"/paper/predicting-head-movement-in-panoramic-video-a","slug":"predicting-head-movement-in-panoramic-video-a","title":"Predicting Head Movement in Panoramic Video: A Deep Reinforcement Learning Approach","date":"2017-10-30","arxiv_id":"1710.10755","repositories_listed":1,"syntology":null},{"url":"/paper/the-effects-of-memory-replay-in-reinforcement","slug":"the-effects-of-memory-replay-in-reinforcement","title":"The Effects of Memory Replay in Reinforcement Learning","date":"2017-10-18","arxiv_id":"1710.06574","repositories_listed":1,"syntology":null},{"url":"/paper/vision-based-deep-execution-monitoring","slug":"vision-based-deep-execution-monitoring","title":"Vision-based deep execution monitoring","date":"2017-09-29","arxiv_id":"1709.10507","repositories_listed":1,"syntology":null},{"url":"/paper/learning-complex-dexterous-manipulation-with","slug":"learning-complex-dexterous-manipulation-with","title":"Learning Complex Dexterous Manipulation with Deep Reinforcement Learning and Demonstrations","date":"2017-09-28","arxiv_id":"1709.10087","repositories_listed":1,"syntology":null},{"url":"/paper/exposure-a-white-box-photo-post-processing","slug":"exposure-a-white-box-photo-post-processing","title":"Exposure: A White-Box Photo Post-Processing Framework","date":"2017-09-27","arxiv_id":"1709.09602","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-event-driven","slug":"deep-reinforcement-learning-for-event-driven","title":"Deep Reinforcement Learning for Event-Driven Multi-Agent Decision Processes","date":"2017-09-19","arxiv_id":"1709.06656","repositories_listed":1,"syntology":null},{"url":"/paper/guided-deep-reinforcement-learning-for-swarm","slug":"guided-deep-reinforcement-learning-for-swarm","title":"Guided Deep Reinforcement Learning for Swarm Systems","date":"2017-09-18","arxiv_id":"1709.06011","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/guided-deep-reinforcement-learning-for-swarm#ran","syntology_url":"https://syntology.ai/paper/1709.06011","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1709.06011"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-for","slug":"deep-reinforcement-learning-for","title":"Deep Reinforcement Learning for Conversational AI","date":"2017-09-15","arxiv_id":"1709.05067","repositories_listed":1,"syntology":null},{"url":"/paper/automated-cloud-provisioning-on-aws-using","slug":"automated-cloud-provisioning-on-aws-using","title":"Automated Cloud Provisioning on AWS using Deep Reinforcement Learning","date":"2017-09-13","arxiv_id":"1709.04305","repositories_listed":1,"syntology":null},{"url":"/paper/prosocial-learning-agents-solve-generalized","slug":"prosocial-learning-agents-solve-generalized","title":"Prosocial learning agents solve generalized Stag Hunts better than selfish ones","date":"2017-09-08","arxiv_id":"1709.02865","repositories_listed":1,"syntology":null},{"url":"/paper/reproducibility-of-benchmarked-deep","slug":"reproducibility-of-benchmarked-deep","title":"Reproducibility of Benchmarked Deep Reinforcement Learning Tasks for Continuous Control","date":"2017-08-10","arxiv_id":"1708.04133","repositories_listed":1,"syntology":null},{"url":"/paper/learning-how-to-active-learn-a-deep","slug":"learning-how-to-active-learn-a-deep","title":"Learning how to Active Learn: A Deep Reinforcement Learning Approach","date":"2017-08-08","arxiv_id":"1708.02383","repositories_listed":1,"syntology":null},{"url":"/paper/grounding-language-for-transfer-in-deep","slug":"grounding-language-for-transfer-in-deep","title":"Grounding Language for Transfer in Deep Reinforcement Learning","date":"2017-08-01","arxiv_id":"1708.00133","repositories_listed":1,"syntology":null}],"record_sha256":"6eeda70e284a275335e3dd503a85d89cb930b14a098ba204f6603ec29b842960","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}