{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/3","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":59,"rows_per_page":100,"rows":[201,300],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/2","next":"/task/deep-reinforcement-learning/papers/4","papers":[{"url":"/paper/coderl-mastering-code-generation-through","slug":"coderl-mastering-code-generation-through","title":"CodeRL: Mastering Code Generation through Pretrained Models and Deep Reinforcement Learning","date":"2022-07-05","arxiv_id":"2207.01780","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/coderl-mastering-code-generation-through#ran","syntology_url":"https://syntology.ai/paper/2207.01780","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.01780"}},"official":{"repos":["salesforce/coderl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generalized-policy-improvement-algorithms","slug":"generalized-policy-improvement-algorithms","title":"Generalized Policy Improvement Algorithms with Theoretically Supported Sample Reuse","date":"2022-06-28","arxiv_id":"2206.13714","repositories_listed":2,"syntology":null},{"url":"/paper/collaborative-target-search-with-a-visual","slug":"collaborative-target-search-with-a-visual","title":"Collaborative Target Search with a Visual Drone Swarm: An Adaptive Curriculum Embedded Multistage Reinforcement Learning Approach","date":"2022-04-26","arxiv_id":"2204.12181","repositories_listed":2,"syntology":null},{"url":"/paper/5g-routing-interfered-environment","slug":"5g-routing-interfered-environment","title":"5G Routing Interfered Environment","date":"2022-03-28","arxiv_id":"2203.14790","repositories_listed":2,"syntology":null},{"url":"/paper/quantum-deep-reinforcement-learning-for-robot","slug":"quantum-deep-reinforcement-learning-for-robot","title":"Quantum Deep Reinforcement Learning for Robot Navigation Tasks","date":"2022-02-24","arxiv_id":"2202.12180","repositories_listed":2,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/quantum-deep-reinforcement-learning-for-robot#ran","syntology_url":"https://syntology.ai/paper/2202.12180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.12180"}},"official":{"repos":["dfki-ric-quantum/qdrl-turtlebot-env","dfki-ric-quantum/qdrl-turtlebot-eval"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/simsr-simple-distance-based-state","slug":"simsr-simple-distance-based-state","title":"SimSR: Simple Distance-based State Representation for Deep Reinforcement Learning","date":"2021-12-31","arxiv_id":"2112.15303","repositories_listed":2,"syntology":null},{"url":"/paper/lane-change-decision-making-through-deep-1","slug":"lane-change-decision-making-through-deep-1","title":"Lane Change Decision-Making through Deep Reinforcement Learning","date":"2021-12-24","arxiv_id":"2112.14705","repositories_listed":2,"syntology":null},{"url":"/paper/cleanrl-high-quality-single-file","slug":"cleanrl-high-quality-single-file","title":"CleanRL: High-quality Single-file Implementations of Deep Reinforcement Learning Algorithms","date":"2021-11-16","arxiv_id":"2111.08819","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cleanrl-high-quality-single-file#ran","syntology_url":"https://syntology.ai/paper/2111.08819","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.08819"}},"official":{"repos":["vwxyzjn/cleanrl"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/d3rlpy-an-offline-deep-reinforcement-learning","slug":"d3rlpy-an-offline-deep-reinforcement-learning","title":"d3rlpy: An Offline Deep Reinforcement Learning Library","date":"2021-11-06","arxiv_id":"2111.03788","repositories_listed":2,"syntology":null},{"url":"/paper/learning-temporally-consistent-1","slug":"learning-temporally-consistent-1","title":"Learning Temporally-Consistent Representations for Data-Efficient Reinforcement Learning","date":"2021-10-11","arxiv_id":"2110.04935","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":2,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-temporally-consistent-1#ran","syntology_url":"https://syntology.ai/paper/2110.04935","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.04935"}},"official":{"repos":["anon-researcher-repo/ksl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-batch-experience-replay","slug":"large-batch-experience-replay","title":"Large Batch Experience Replay","date":"2021-10-04","arxiv_id":"2110.01528","repositories_listed":2,"syntology":{"n":11,"n_ran":7,"n_constructed":4,"n_ran_checked":4,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/large-batch-experience-replay#ran","syntology_url":"https://syntology.ai/paper/2110.01528","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.01528"}},"official":{"repos":["sureli/laber"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/real-robot-challenge-using-deep-reinforcement","slug":"real-robot-challenge-using-deep-reinforcement","title":"Solving the Real Robot Challenge using Deep Reinforcement Learning","date":"2021-09-30","arxiv_id":"2109.15233","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/real-robot-challenge-using-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2109.15233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2109.15233"}},"official":{"repos":["robertmccarthy97/rrc_phase1"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-practically-feasible-policies-for","slug":"learning-practically-feasible-policies-for","title":"Learning Practically Feasible Policies for Online 3D Bin Packing","date":"2021-08-31","arxiv_id":"2108.13680","repositories_listed":2,"syntology":null},{"url":"/paper/marsexplorer-exploration-of-unknown-terrains","slug":"marsexplorer-exploration-of-unknown-terrains","title":"MarsExplorer: Exploration of Unknown Terrains via Deep Reinforcement Learning and Procedurally Generated Environments","date":"2021-07-21","arxiv_id":"2107.09996","repositories_listed":2,"syntology":null},{"url":"/paper/optical-tactile-sim-to-real-policy-transfer","slug":"optical-tactile-sim-to-real-policy-transfer","title":"Tactile Sim-to-Real Policy Transfer via Real-to-Sim Image Translation","date":"2021-06-16","arxiv_id":"2106.08796","repositories_listed":2,"syntology":null},{"url":"/paper/playvirtual-augmenting-cycle-consistent","slug":"playvirtual-augmenting-cycle-consistent","title":"PlayVirtual: Augmenting Cycle-Consistent Virtual Trajectories for Reinforcement Learning","date":"2021-06-08","arxiv_id":"2106.04152","repositories_listed":2,"syntology":{"n":9,"n_ran":7,"n_constructed":4,"n_ran_checked":4,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/playvirtual-augmenting-cycle-consistent#ran","syntology_url":"https://syntology.ai/paper/2106.04152","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.04152"}},"official":{"repos":["microsoft/Playvirtual"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mico-learning-improved-representations-via","slug":"mico-learning-improved-representations-via","title":"MICo: Improved representations via sampling-based state similarity for Markov decision processes","date":"2021-06-03","arxiv_id":"2106.08229","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":1,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 1 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mico-learning-improved-representations-via#ran","syntology_url":"https://syntology.ai/paper/2106.08229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.08229"}},"official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/action-advising-with-advice-imitation-in-deep","slug":"action-advising-with-advice-imitation-in-deep","title":"Action Advising with Advice Imitation in Deep Reinforcement Learning","date":"2021-04-17","arxiv_id":"2104.08441","repositories_listed":2,"syntology":null},{"url":"/paper/improving-robustness-of-deep-reinforcement","slug":"improving-robustness-of-deep-reinforcement","title":"Improving Robustness of Deep Reinforcement Learning Agents: Environment Attack based on the Critic Network","date":"2021-04-07","arxiv_id":"2104.03154","repositories_listed":2,"syntology":null},{"url":"/paper/discovering-diverse-solutions-in-deep","slug":"discovering-diverse-solutions-in-deep","title":"Discovering Diverse Solutions in Deep Reinforcement Learning by Maximizing State-Action-Based Mutual Information","date":"2021-03-12","arxiv_id":"2103.07084","repositories_listed":2,"syntology":null},{"url":"/paper/continuous-coordination-as-a-realistic","slug":"continuous-coordination-as-a-realistic","title":"Continuous Coordination As a Realistic Scenario for Lifelong Learning","date":"2021-03-04","arxiv_id":"2103.03216","repositories_listed":2,"syntology":null},{"url":"/paper/decoupling-value-and-policy-for","slug":"decoupling-value-and-policy-for","title":"Decoupling Value and Policy for Generalization in Reinforcement Learning","date":"2021-02-20","arxiv_id":"2102.10330","repositories_listed":2,"syntology":null},{"url":"/paper/state-entropy-maximization-with-random","slug":"state-entropy-maximization-with-random","title":"State Entropy Maximization with Random Encoders for Efficient Exploration","date":"2021-02-18","arxiv_id":"2102.09430","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-dynamic","slug":"deep-reinforcement-learning-with-dynamic","title":"Tactical Optimism and Pessimism for Deep Reinforcement Learning","date":"2021-02-07","arxiv_id":"2102.03765","repositories_listed":2,"syntology":null},{"url":"/paper/counterfactual-state-explanations-for","slug":"counterfactual-state-explanations-for","title":"Counterfactual State Explanations for Reinforcement Learning Agents via Generative Deep Learning","date":"2021-01-29","arxiv_id":"2101.12446","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/counterfactual-state-explanations-for#ran","syntology_url":"https://syntology.ai/paper/2101.12446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.12446"}},"official":{"repos":["mattolson93/counterfactual-state-explanations"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/robust-reinforcement-learning-on-state-1","slug":"robust-reinforcement-learning-on-state-1","title":"Robust Reinforcement Learning on State Observations with Learned Optimal Adversary","date":"2021-01-21","arxiv_id":"2101.08452","repositories_listed":2,"syntology":{"n":8,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/robust-reinforcement-learning-on-state-1#ran","syntology_url":"https://syntology.ai/paper/2101.08452","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.08452"}},"official":{"repos":["huanzhang12/ATLA_robust_RL"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/solving-common-payoff-games-with-approximate","slug":"solving-common-payoff-games-with-approximate","title":"Solving Common-Payoff Games with Approximate Policy Iteration","date":"2021-01-11","arxiv_id":"2101.04237","repositories_listed":2,"syntology":{"n":9,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/solving-common-payoff-games-with-approximate#ran","syntology_url":"https://syntology.ai/paper/2101.04237","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.04237"}},"official":{"repos":["ssokota/capi","ssokota/tiny-hanabi"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-of-graph-matching","slug":"deep-reinforcement-learning-of-graph-matching","title":"Revocable Deep Reinforcement Learning with Affinity Regularization for Outlier-Robust Graph Matching","date":"2020-12-16","arxiv_id":"2012.08950","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":1,"n_instrument":6,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-reinforcement-learning-of-graph-matching#ran","syntology_url":"https://syntology.ai/paper/2012.08950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.08950"}},"official":{"repos":["thinklab-sjtu/rgm","Thinklab-SJTU/awesome-ml4co"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/bebold-exploration-beyond-the-boundary-of-1","slug":"bebold-exploration-beyond-the-boundary-of-1","title":"BeBold: Exploration Beyond the Boundary of Explored Regions","date":"2020-12-15","arxiv_id":"2012.08621","repositories_listed":2,"syntology":null},{"url":"/paper/language-as-a-cognitive-tool-to-imagine-goals-1","slug":"language-as-a-cognitive-tool-to-imagine-goals-1","title":"Language as a Cognitive Tool to Imagine Goals in Curiosity Driven Exploration","date":"2020-12-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/revisiting-rainbow-promoting-more-insightful","slug":"revisiting-rainbow-promoting-more-insightful","title":"Revisiting Rainbow: Promoting more Insightful and Inclusive Deep Reinforcement Learning Research","date":"2020-11-20","arxiv_id":"2011.14826","repositories_listed":2,"syntology":null},{"url":"/paper/softgym-benchmarking-deep-reinforcement","slug":"softgym-benchmarking-deep-reinforcement","title":"SoftGym: Benchmarking Deep Reinforcement Learning for Deformable Object Manipulation","date":"2020-11-14","arxiv_id":"2011.07215","repositories_listed":2,"syntology":null},{"url":"/paper/decentralized-structural-rnn-for-robot-crowd","slug":"decentralized-structural-rnn-for-robot-crowd","title":"Decentralized Structural-RNN for Robot Crowd Navigation with Deep Reinforcement Learning","date":"2020-11-09","arxiv_id":"2011.04820","repositories_listed":2,"syntology":null},{"url":"/paper/learning-guidance-rewards-with-trajectory","slug":"learning-guidance-rewards-with-trajectory","title":"Learning Guidance Rewards with Trajectory-space Smoothing","date":"2020-10-23","arxiv_id":"2010.12718","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-guidance-rewards-with-trajectory#ran","syntology_url":"https://syntology.ai/paper/2010.12718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.12718"}},"official":{"repos":["tgangwani/GuidanceRewards"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/improving-generalization-in-reinforcement","slug":"improving-generalization-in-reinforcement","title":"Improving Generalization in Reinforcement Learning with Mixture Regularization","date":"2020-10-21","arxiv_id":"2010.10814","repositories_listed":2,"syntology":null},{"url":"/paper/visual-navigation-in-real-world-indoor","slug":"visual-navigation-in-real-world-indoor","title":"Visual Navigation in Real-World Indoor Environments Using End-to-End Deep Reinforcement Learning","date":"2020-10-21","arxiv_id":"2010.10903","repositories_listed":2,"syntology":null},{"url":"/paper/uav-path-planning-using-global-and-local-map","slug":"uav-path-planning-using-global-and-local-map","title":"UAV Path Planning using Global and Local Map Information with Deep Reinforcement Learning","date":"2020-10-14","arxiv_id":"2010.06917","repositories_listed":2,"syntology":null},{"url":"/paper/epidemioptim-a-toolbox-for-the-optimization-1","slug":"epidemioptim-a-toolbox-for-the-optimization-1","title":"EpidemiOptim: A Toolbox for the Optimization of Control Policies in Epidemiological Models","date":"2020-10-09","arxiv_id":"2010.04452","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-process","slug":"deep-reinforcement-learning-for-process","title":"Deep Reinforcement Learning for Process Synthesis","date":"2020-09-23","arxiv_id":"2009.13265","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-unknown","slug":"deep-reinforcement-learning-for-unknown","title":"Toward Deep Supervised Anomaly Detection: Reinforcement Learning from Partially Labeled Anomaly Data","date":"2020-09-15","arxiv_id":"2009.06847","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-for-unknown#ran","syntology_url":"https://syntology.ai/paper/2009.06847","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.06847"}},"official":null}},{"url":"/paper/congested-urban-networks-tend-to-be","slug":"congested-urban-networks-tend-to-be","title":"Congested Urban Networks Tend to Be Insensitive to Signal Settings: Implications for Learning-Based Control","date":"2020-08-21","arxiv_id":"2008.10989","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-with-quantum","slug":"reinforcement-learning-with-quantum","title":"Reinforcement Learning with Quantum Variational Circuits","date":"2020-08-15","arxiv_id":"2008.07524","repositories_listed":2,"syntology":null},{"url":"/paper/trifinger-an-open-source-robot-for-learning","slug":"trifinger-an-open-source-robot-for-learning","title":"TriFinger: An Open-Source Robot for Learning Dexterity","date":"2020-08-08","arxiv_id":"2008.03596","repositories_listed":2,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/trifinger-an-open-source-robot-for-learning#ran","syntology_url":"https://syntology.ai/paper/2008.03596","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.03596"}},"official":null}},{"url":"/paper/robust-deep-reinforcement-learning-through","slug":"robust-deep-reinforcement-learning-through","title":"Robust Deep Reinforcement Learning through Adversarial Loss","date":"2020-08-05","arxiv_id":"2008.01976","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-through#ran","syntology_url":"https://syntology.ai/paper/2008.01976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2008.01976"}},"official":{"repos":["tuomaso/radial_rl","tuomaso/radial_rl_v2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/revisiting-fundamentals-of-experience-replay","slug":"revisiting-fundamentals-of-experience-replay","title":"Revisiting Fundamentals of Experience Replay","date":"2020-07-13","arxiv_id":"2007.06700","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/revisiting-fundamentals-of-experience-replay#ran","syntology_url":"https://syntology.ai/paper/2007.06700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.06700"}},"official":{"repos":["google-research/google-research"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/an-equivalence-between-loss-functions-and-non","slug":"an-equivalence-between-loss-functions-and-non","title":"An Equivalence between Loss Functions and Non-Uniform Sampling in Experience Replay","date":"2020-07-12","arxiv_id":"2007.06049","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/an-equivalence-between-loss-functions-and-non#ran","syntology_url":"https://syntology.ai/paper/2007.06049","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2007.06049"}},"official":{"repos":["sfujim/LAP-PAL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/relation-aware-transformer-for-portfolio","slug":"relation-aware-transformer-for-portfolio","title":"Relation-Aware Transformer for Portfolio Policy Learning","date":"2020-07-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/mdp-homomorphic-networks-group-symmetries-in","slug":"mdp-homomorphic-networks-group-symmetries-in","title":"MDP Homomorphic Networks: Group Symmetries in Reinforcement Learning","date":"2020-06-30","arxiv_id":"2006.16908","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mdp-homomorphic-networks-group-symmetries-in#ran","syntology_url":"https://syntology.ai/paper/2006.16908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.16908"}},"official":{"repos":["ElisevanderPol/mdp-homomorphic-networks"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-closer-look-at-invalid-action-masking-in","slug":"a-closer-look-at-invalid-action-masking-in","title":"A Closer Look at Invalid Action Masking in Policy Gradient Algorithms","date":"2020-06-25","arxiv_id":"2006.14171","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-closer-look-at-invalid-action-masking-in#ran","syntology_url":"https://syntology.ai/paper/2006.14171","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.14171"}},"official":{"repos":["vwxyzjn/invalid-action-masking"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/pipeline-psro-a-scalable-approach-for-finding","slug":"pipeline-psro-a-scalable-approach-for-finding","title":"Pipeline PSRO: A Scalable Approach for Finding Approximate Nash Equilibria in Large Games","date":"2020-06-15","arxiv_id":"2006.08555","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pipeline-psro-a-scalable-approach-for-finding#ran","syntology_url":"https://syntology.ai/paper/2006.08555","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.08555"}},"official":{"repos":["JBLanier/pipeline-psro"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-play-no-press-diplomacy-with-best","slug":"learning-to-play-no-press-diplomacy-with-best","title":"Learning to Play No-Press Diplomacy with Best Response Policy Iteration","date":"2020-06-08","arxiv_id":"2006.04635","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-play-no-press-diplomacy-with-best#ran","syntology_url":"https://syntology.ai/paper/2006.04635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2006.04635"}},"official":{"repos":["deepmind/diplomacy"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/implementation-matters-in-deep-rl-a-case","slug":"implementation-matters-in-deep-rl-a-case","title":"Implementation Matters in Deep RL: A Case Study on PPO and TRPO","date":"2020-05-01","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/chip-placement-with-deep-reinforcement","slug":"chip-placement-with-deep-reinforcement","title":"Chip Placement with Deep Reinforcement Learning","date":"2020-04-22","arxiv_id":"2004.10746","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chip-placement-with-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2004.10746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2004.10746"}},"official":null}},{"url":"/paper/modeling-3d-shapes-by-reinforcement-learning","slug":"modeling-3d-shapes-by-reinforcement-learning","title":"Modeling 3D Shapes by Reinforcement Learning","date":"2020-03-27","arxiv_id":"2003.12397","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/modeling-3d-shapes-by-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2003.12397","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.12397"}},"official":null}},{"url":"/paper/exploring-unknown-states-with-action-balance","slug":"exploring-unknown-states-with-action-balance","title":"Exploring Unknown States with Action Balance","date":"2020-03-10","arxiv_id":"2003.04518","repositories_listed":2,"syntology":null},{"url":"/paper/learning-when-and-where-to-zoom-with-deep","slug":"learning-when-and-where-to-zoom-with-deep","title":"Learning When and Where to Zoom with Deep Reinforcement Learning","date":"2020-03-01","arxiv_id":"2003.00425","repositories_listed":2,"syntology":null},{"url":"/paper/language-as-a-cognitive-tool-to-imagine-goals","slug":"language-as-a-cognitive-tool-to-imagine-goals","title":"Language as a Cognitive Tool to Imagine Goals in Curiosity-Driven Exploration","date":"2020-02-21","arxiv_id":"2002.09253","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/language-as-a-cognitive-tool-to-imagine-goals#ran","syntology_url":"https://syntology.ai/paper/2002.09253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2002.09253"}},"official":{"repos":["flowersteam/Imagine","flowersteam/playground_env"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/soft-hindsight-experience-replay","slug":"soft-hindsight-experience-replay","title":"Soft Hindsight Experience Replay","date":"2020-02-06","arxiv_id":"2002.02089","repositories_listed":2,"syntology":null},{"url":"/paper/multitask-radiological-modality-invariant","slug":"multitask-radiological-modality-invariant","title":"Multitask radiological modality invariant landmark localization using deep reinforcement learning","date":"2020-01-25","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/audio-visual-embodied-navigation","slug":"audio-visual-embodied-navigation","title":"SoundSpaces: Audio-Visual Navigation in 3D Environments","date":"2019-12-24","arxiv_id":"1912.11474","repositories_listed":2,"syntology":null},{"url":"/paper/explain-your-move-understanding-agent-actions-1","slug":"explain-your-move-understanding-agent-actions-1","title":"Explain Your Move: Understanding Agent Actions Using Specific and Relevant Feature Attribution","date":"2019-12-23","arxiv_id":"1912.12191","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":3,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 3 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/explain-your-move-understanding-agent-actions-1#ran","syntology_url":"https://syntology.ai/paper/1912.12191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.12191"}},"official":{"repos":["rl-interpretation/understandingRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"url":"/paper/pic-permutation-invariant-critic-for-multi","slug":"pic-permutation-invariant-critic-for-multi","title":"PIC: Permutation Invariant Critic for Multi-Agent Deep Reinforcement Learning","date":"2019-10-31","arxiv_id":"1911.00025","repositories_listed":2,"syntology":null},{"url":"/paper/zpd-teaching-strategies-for-deep","slug":"zpd-teaching-strategies-for-deep","title":"ZPD Teaching Strategies for Deep Reinforcement Learning from Demonstrations","date":"2019-10-26","arxiv_id":"1910.12154","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/zpd-teaching-strategies-for-deep#ran","syntology_url":"https://syntology.ai/paper/1910.12154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.12154"}},"official":null}},{"url":"/paper/regularization-matters-in-policy-optimization-1","slug":"regularization-matters-in-policy-optimization-1","title":"Regularization Matters in Policy Optimization","date":"2019-10-21","arxiv_id":"1910.09191","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regularization-matters-in-policy-optimization-1#ran","syntology_url":"https://syntology.ai/paper/1910.09191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.09191"}},"official":{"repos":["xuanlinli17/iclr2021_rlreg","xuanlinli17/po-rl-regularization"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/teacher-algorithms-for-curriculum-learning-of","slug":"teacher-algorithms-for-curriculum-learning-of","title":"Teacher algorithms for curriculum learning of Deep RL in continuously parameterized environments","date":"2019-10-16","arxiv_id":"1910.07224","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/teacher-algorithms-for-curriculum-learning-of#ran","syntology_url":"https://syntology.ai/paper/1910.07224","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1910.07224"}},"official":{"repos":["flowersteam/teachDeepRL"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/modular-deep-reinforcement-learning-with","slug":"modular-deep-reinforcement-learning-with","title":"Modular Deep Reinforcement Learning with Temporal Logic Specifications","date":"2019-09-23","arxiv_id":"1909.11591","repositories_listed":2,"syntology":null},{"url":"/paper/driving-in-dense-traffic-with-model-free","slug":"driving-in-dense-traffic-with-model-free","title":"Driving in Dense Traffic with Model-Free Reinforcement Learning","date":"2019-09-15","arxiv_id":"1909.06710","repositories_listed":2,"syntology":null},{"url":"/paper/flight-controller-synthesis-via-deep","slug":"flight-controller-synthesis-via-deep","title":"Flight Controller Synthesis Via Deep Reinforcement Learning","date":"2019-09-14","arxiv_id":"1909.06493","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-algorithm-for","slug":"deep-reinforcement-learning-algorithm-for","title":"Deep Reinforcement Learning Algorithm for Dynamic Pricing of Express Lanes with Multiple Access Locations","date":"2019-09-10","arxiv_id":"1909.04760","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-control-of","slug":"deep-reinforcement-learning-for-control-of","title":"Deep Reinforcement Learning for Control of Probabilistic Boolean Networks","date":"2019-09-07","arxiv_id":"1909.03331","repositories_listed":2,"syntology":null},{"url":"/paper/classification-with-costly-features-as-a","slug":"classification-with-costly-features-as-a","title":"Classification with Costly Features as a Sequential Decision-Making Problem","date":"2019-09-05","arxiv_id":"1909.02564","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/classification-with-costly-features-as-a#ran","syntology_url":"https://syntology.ai/paper/1909.02564","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1909.02564"}},"official":{"repos":["jaromiru/cwcf"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/performing-deep-recurrent-double-q-learning","slug":"performing-deep-recurrent-double-q-learning","title":"Performing Deep Recurrent Double Q-Learning for Atari Games","date":"2019-08-16","arxiv_id":"1908.06040","repositories_listed":2,"syntology":null},{"url":"/paper/modern-deep-reinforcement-learning-algorithms","slug":"modern-deep-reinforcement-learning-algorithms","title":"Modern Deep Reinforcement Learning Algorithms","date":"2019-06-24","arxiv_id":"1906.10025","repositories_listed":2,"syntology":null},{"url":"/paper/language-as-an-abstraction-for-hierarchical","slug":"language-as-an-abstraction-for-hierarchical","title":"Language as an Abstraction for Hierarchical Deep Reinforcement Learning","date":"2019-06-18","arxiv_id":"1906.07343","repositories_listed":2,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/language-as-an-abstraction-for-hierarchical#ran","syntology_url":"https://syntology.ai/paper/1906.07343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.07343"}},"official":{"repos":["google-research/clevr_robot_env"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cross-view-policy-learning-for-street","slug":"cross-view-policy-learning-for-street","title":"Cross-View Policy Learning for Street Navigation","date":"2019-06-13","arxiv_id":"1906.05930","repositories_listed":2,"syntology":null},{"url":"/paper/adversarial-policies-attacking-deep","slug":"adversarial-policies-attacking-deep","title":"Adversarial Policies: Attacking Deep Reinforcement Learning","date":"2019-05-25","arxiv_id":"1905.10615","repositories_listed":2,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/adversarial-policies-attacking-deep#ran","syntology_url":"https://syntology.ai/paper/1905.10615","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.10615"}},"official":{"repos":["HumanCompatibleAI/adversarial-policies"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/estimating-risk-and-uncertainty-in-deep","slug":"estimating-risk-and-uncertainty-in-deep","title":"Estimating Risk and Uncertainty in Deep Reinforcement Learning","date":"2019-05-23","arxiv_id":"1905.09638","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/estimating-risk-and-uncertainty-in-deep#ran","syntology_url":"https://syntology.ai/paper/1905.09638","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.09638"}},"official":{"repos":["IndustAI/risk-and-uncertainty"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cobra-data-efficient-model-based-rl-through","slug":"cobra-data-efficient-model-based-rl-through","title":"COBRA: Data-Efficient Model-Based RL through Unsupervised Object Discovery and Curiosity-Driven Exploration","date":"2019-05-22","arxiv_id":"1905.09275","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cobra-data-efficient-model-based-rl-through#ran","syntology_url":"https://syntology.ai/paper/1905.09275","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.09275"}},"official":{"repos":["deepmind/spriteworld"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-reinforcement-learning-with-task","slug":"meta-reinforcement-learning-with-task","title":"Meta Reinforcement Learning with Task Embedding and Shared Policy","date":"2019-05-16","arxiv_id":"1905.06527","repositories_listed":2,"syntology":null},{"url":"/paper/trajectory-based-off-policy-deep","slug":"trajectory-based-off-policy-deep","title":"Trajectory-Based Off-Policy Deep Reinforcement Learning","date":"2019-05-14","arxiv_id":"1905.05710","repositories_listed":2,"syntology":null},{"url":"/paper/toybox-a-suite-of-environments-for","slug":"toybox-a-suite-of-environments-for","title":"Toybox: A Suite of Environments for Experimental Evaluation of Deep Reinforcement Learning","date":"2019-05-07","arxiv_id":"1905.02825","repositories_listed":2,"syntology":null},{"url":"/paper/synthesis-of-biologically-realistic-human","slug":"synthesis-of-biologically-realistic-human","title":"Synthesis of Biologically Realistic Human Motion Using Joint Torque Actuation","date":"2019-04-30","arxiv_id":"1904.13041","repositories_listed":2,"syntology":null},{"url":"/paper/model-free-deep-reinforcement-learning-for","slug":"model-free-deep-reinforcement-learning-for","title":"Model-free Deep Reinforcement Learning for Urban Autonomous Driving","date":"2019-04-20","arxiv_id":"1904.09503","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-feedback","slug":"deep-reinforcement-learning-with-feedback","title":"Deep Reinforcement Learning with Feedback-based Exploration","date":"2019-03-14","arxiv_id":"1903.06151","repositories_listed":2,"syntology":null},{"url":"/paper/ros2learn-a-reinforcement-learning-framework","slug":"ros2learn-a-reinforcement-learning-framework","title":"ROS2Learn: a reinforcement learning framework for ROS 2","date":"2019-03-14","arxiv_id":"1903.06282","repositories_listed":2,"syntology":null},{"url":"/paper/learning-heuristics-over-large-graphs-via","slug":"learning-heuristics-over-large-graphs-via","title":"Learning Heuristics over Large Graphs via Deep Reinforcement Learning","date":"2019-03-08","arxiv_id":"1903.03332","repositories_listed":2,"syntology":null},{"url":"/paper/learning-to-follow-directions-in-street-view","slug":"learning-to-follow-directions-in-street-view","title":"Learning To Follow Directions in Street View","date":"2019-03-01","arxiv_id":"1903.00401","repositories_listed":2,"syntology":null},{"url":"/paper/trojdrl-trojan-attacks-on-deep-reinforcement","slug":"trojdrl-trojan-attacks-on-deep-reinforcement","title":"TrojDRL: Trojan Attacks on Deep Reinforcement Learning Agents","date":"2019-03-01","arxiv_id":"1903.06638","repositories_listed":2,"syntology":null},{"url":"/paper/190504100","slug":"190504100","title":"Deep Reinforcement Learning using Genetic Algorithm for Parameter Optimization","date":"2019-02-19","arxiv_id":"1905.04100","repositories_listed":2,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":7,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/190504100#ran","syntology_url":"https://syntology.ai/paper/1905.04100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1905.04100"}},"official":{"repos":["aralab-unr/ReinforcementLearningWithGA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/trust-region-guided-proximal-policy","slug":"trust-region-guided-proximal-policy","title":"Trust Region-Guided Proximal Policy Optimization","date":"2019-01-29","arxiv_id":"1901.10314","repositories_listed":2,"syntology":null},{"url":"/paper/action-robust-reinforcement-learning-and","slug":"action-robust-reinforcement-learning-and","title":"Action Robust Reinforcement Learning and Applications in Continuous Control","date":"2019-01-26","arxiv_id":"1901.09184","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/action-robust-reinforcement-learning-and#ran","syntology_url":"https://syntology.ai/paper/1901.09184","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.09184"}},"official":{"repos":["tesslerc/ActionRobustRL","icml2019-anonymous-author/Action-Robust-Reinforcement-Learning"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-policy-trust-region-policy-optimisation","slug":"on-policy-trust-region-policy-optimisation","title":"On-Policy Trust Region Policy Optimisation with Replay Buffers","date":"2019-01-18","arxiv_id":"1901.06212","repositories_listed":2,"syntology":null},{"url":"/paper/energy-efficient-thermal-comfort-control-in","slug":"energy-efficient-thermal-comfort-control-in","title":"Energy-Efficient Thermal Comfort Control in Smart Buildings via Deep Reinforcement Learning","date":"2019-01-15","arxiv_id":"1901.04693","repositories_listed":2,"syntology":null},{"url":"/paper/decentralized-computation-offloading-for","slug":"decentralized-computation-offloading-for","title":"Decentralized Computation Offloading for Multi-User Mobile Edge Computing: A Deep Reinforcement Learning Approach","date":"2018-12-16","arxiv_id":"1812.07394","repositories_listed":2,"syntology":null},{"url":"/paper/improving-automatic-source-code-summarization","slug":"improving-automatic-source-code-summarization","title":"Improving Automatic Source Code Summarization via Deep Reinforcement Learning","date":"2018-11-17","arxiv_id":"1811.07234","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/improving-automatic-source-code-summarization#ran","syntology_url":"https://syntology.ai/paper/1811.07234","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1811.07234"}},"official":null}},{"url":"/paper/reward-learning-from-human-preferences-and","slug":"reward-learning-from-human-preferences-and","title":"Reward learning from human preferences and demonstrations in Atari","date":"2018-11-15","arxiv_id":"1811.06521","repositories_listed":2,"syntology":null},{"url":"/paper/memory-based-deep-reinforcement-learning-for","slug":"memory-based-deep-reinforcement-learning-for","title":"Memory-based Deep Reinforcement Learning for Obstacle Avoidance in UAV with Limited Environment Knowledge","date":"2018-11-08","arxiv_id":"1811.03307","repositories_listed":2,"syntology":null},{"url":"/paper/making-sense-of-vision-and-touch-self","slug":"making-sense-of-vision-and-touch-self","title":"Making Sense of Vision and Touch: Self-Supervised Learning of Multimodal Representations for Contact-Rich Tasks","date":"2018-10-24","arxiv_id":"1810.10191","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/making-sense-of-vision-and-touch-self#ran","syntology_url":"https://syntology.ai/paper/1810.10191","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.10191"}},"official":{"repos":["stanford-iprl-lab/multimodal_representation"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-deep-reinforcement-learning-using-online","slug":"fast-deep-reinforcement-learning-using-online","title":"Fast deep reinforcement learning using online adjustments from the past","date":"2018-10-18","arxiv_id":"1810.08163","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/fast-deep-reinforcement-learning-using-online#ran","syntology_url":"https://syntology.ai/paper/1810.08163","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.08163"}},"official":null}},{"url":"/paper/cem-rl-combining-evolutionary-and-gradient-1","slug":"cem-rl-combining-evolutionary-and-gradient-1","title":"CEM-RL: Combining evolutionary and gradient-based methods for policy search","date":"2018-10-02","arxiv_id":"1810.01222","repositories_listed":2,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"0 ran · 2 unverified","sample_list":"/paper/cem-rl-combining-evolutionary-and-gradient-1#ran","syntology_url":"https://syntology.ai/paper/1810.01222","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.01222"}},"official":{"repos":["apourchot/CEM-RL"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}}],"record_sha256":"ca63e2f798208d590d778aa35674b811c05a1ad73943f61c16fbf3a4bdc051a5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}