{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/2","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":2,"pages_in_order":59,"rows_per_page":100,"rows":[101,200],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning","next":"/task/deep-reinforcement-learning/papers/3","papers":[{"url":"/paper/deep-bayesian-bandits-showdown-an-empirical","slug":"deep-bayesian-bandits-showdown-an-empirical","title":"Deep Bayesian Bandits Showdown: An Empirical Comparison of Bayesian Deep Networks for Thompson Sampling","date":"2018-02-26","arxiv_id":"1802.09127","repositories_listed":4,"syntology":null},{"url":"/paper/whatever-does-not-kill-deep-reinforcement","slug":"whatever-does-not-kill-deep-reinforcement","title":"Whatever Does Not Kill Deep Reinforcement Learning, Makes It Stronger","date":"2017-12-23","arxiv_id":"1712.09344","repositories_listed":4,"syntology":null},{"url":"/paper/a-deeper-look-at-experience-replay","slug":"a-deeper-look-at-experience-replay","title":"A Deeper Look at Experience Replay","date":"2017-12-04","arxiv_id":"1712.01275","repositories_listed":4,"syntology":null},{"url":"/paper/deep-reinforcement-learning-that-matters","slug":"deep-reinforcement-learning-that-matters","title":"Deep Reinforcement Learning that Matters","date":"2017-09-19","arxiv_id":"1709.06560","repositories_listed":4,"syntology":null},{"url":"/paper/leveraging-demonstrations-for-deep","slug":"leveraging-demonstrations-for-deep","title":"Leveraging Demonstrations for Deep Reinforcement Learning on Robotics Problems with Sparse Rewards","date":"2017-07-27","arxiv_id":"1707.08817","repositories_listed":4,"syntology":null},{"url":"/paper/a-multi-agent-reinforcement-learning-model-of","slug":"a-multi-agent-reinforcement-learning-model-of","title":"A multi-agent reinforcement learning model of common-pool resource appropriation","date":"2017-07-20","arxiv_id":"1707.06600","repositories_listed":4,"syntology":null},{"url":"/paper/thinking-fast-and-slow-with-deep-learning-and","slug":"thinking-fast-and-slow-with-deep-learning-and","title":"Thinking Fast and Slow with Deep Learning and Tree Search","date":"2017-05-23","arxiv_id":"1705.08439","repositories_listed":4,"syntology":null},{"url":"/paper/neural-episodic-control","slug":"neural-episodic-control","title":"Neural Episodic Control","date":"2017-03-06","arxiv_id":"1703.01988","repositories_listed":4,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/neural-episodic-control#ran","syntology_url":"https://syntology.ai/paper/1703.01988","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1703.01988"}},"official":null}},{"url":"/paper/hierarchical-deep-reinforcement-learning","slug":"hierarchical-deep-reinforcement-learning","title":"Hierarchical Deep Reinforcement Learning: Integrating Temporal Abstraction and Intrinsic Motivation","date":"2016-04-20","arxiv_id":"1604.06057","repositories_listed":4,"syntology":null},{"url":"/paper/multiagent-cooperation-and-competition-with","slug":"multiagent-cooperation-and-competition-with","title":"Multiagent Cooperation and Competition with Deep Reinforcement Learning","date":"2015-11-27","arxiv_id":"1511.08779","repositories_listed":4,"syntology":null},{"url":"/paper/beyond-the-rainbow-high-performance-deep","slug":"beyond-the-rainbow-high-performance-deep","title":"Beyond The Rainbow: High Performance Deep Reinforcement Learning on a Desktop PC","date":"2024-11-06","arxiv_id":"2411.03820","repositories_listed":3,"syntology":{"n":25,"n_ran":16,"n_constructed":10,"n_ran_checked":13,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":21,"phrase":"16 ran (of which 10 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/beyond-the-rainbow-high-performance-deep#ran","syntology_url":"https://syntology.ai/paper/2411.03820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03820"}},"official":{"repos":["viptankz/btr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/rl4co-an-extensive-reinforcement-learning-for","slug":"rl4co-an-extensive-reinforcement-learning-for","title":"RL4CO: an Extensive Reinforcement Learning for Combinatorial Optimization Benchmark","date":"2023-06-29","arxiv_id":"2306.17100","repositories_listed":3,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/rl4co-an-extensive-reinforcement-learning-for#ran","syntology_url":"https://syntology.ai/paper/2306.17100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.17100"}},"official":{"repos":["ai4co/rl4co","pytorch/rl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/the-dormant-neuron-phenomenon-in-deep","slug":"the-dormant-neuron-phenomenon-in-deep","title":"The Dormant Neuron Phenomenon in Deep Reinforcement Learning","date":"2023-02-24","arxiv_id":"2302.12902","repositories_listed":3,"syntology":{"n":8,"n_ran":4,"n_constructed":1,"n_ran_checked":1,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":8,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/the-dormant-neuron-phenomenon-in-deep#ran","syntology_url":"https://syntology.ai/paper/2302.12902","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12902"}},"official":{"repos":["google/dopamine"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/learning-low-frequency-motion-control-for","slug":"learning-low-frequency-motion-control-for","title":"Learning Low-Frequency Motion Control for Robust and Dynamic Robot Locomotion","date":"2022-09-29","arxiv_id":"2209.14887","repositories_listed":3,"syntology":null},{"url":"/paper/mlink-linking-black-box-models-from-multiple","slug":"mlink-linking-black-box-models-from-multiple","title":"MLink: Linking Black-Box Models from Multiple Domains for Collaborative Inference","date":"2022-09-28","arxiv_id":"2209.13883","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-multi-agent-2","slug":"deep-reinforcement-learning-for-multi-agent-2","title":"Deep Reinforcement Learning for Multi-Agent Interaction","date":"2022-08-02","arxiv_id":"2208.01769","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-turbulence","slug":"deep-reinforcement-learning-for-turbulence","title":"Deep Reinforcement Learning for Turbulence Modeling in Large Eddy Simulations","date":"2022-06-21","arxiv_id":"2206.11038","repositories_listed":3,"syntology":null},{"url":"/paper/a-unified-approach-to-reinforcement-learning","slug":"a-unified-approach-to-reinforcement-learning","title":"A Unified Approach to Reinforcement Learning, Quantal Response Equilibria, and Two-Player Zero-Sum Games","date":"2022-06-12","arxiv_id":"2206.05825","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":3,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 3 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-unified-approach-to-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2206.05825","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.05825"}},"official":{"repos":["deepmind/open_spiel"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]}}},{"url":"/paper/warpdrive-extremely-fast-end-to-end-deep","slug":"warpdrive-extremely-fast-end-to-end-deep","title":"WarpDrive: Extremely Fast End-to-End Deep Multi-Agent Reinforcement Learning on a GPU","date":"2021-08-31","arxiv_id":"2108.13976","repositories_listed":3,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/warpdrive-extremely-fast-end-to-end-deep#ran","syntology_url":"https://syntology.ai/paper/2108.13976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.13976"}},"official":{"repos":["salesforce/warp-drive"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deep-reinforcement-learning-at-the-edge-of","slug":"deep-reinforcement-learning-at-the-edge-of","title":"Deep Reinforcement Learning at the Edge of the Statistical Precipice","date":"2021-08-30","arxiv_id":"2108.13264","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":4,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 4 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-at-the-edge-of#ran","syntology_url":"https://syntology.ai/paper/2108.13264","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2108.13264"}},"official":{"repos":["google-research/rliable"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/meta-variationally-intrinsic-motivated","slug":"meta-variationally-intrinsic-motivated","title":"MetaVIM: Meta Variationally Intrinsic Motivated Reinforcement Learning for Decentralized Traffic Signal Control","date":"2021-01-04","arxiv_id":"2101.00746","repositories_listed":3,"syntology":null},{"url":"/paper/decoupling-representation-learning-from","slug":"decoupling-representation-learning-from","title":"Decoupling Representation Learning from Reinforcement Learning","date":"2020-09-14","arxiv_id":"2009.08319","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decoupling-representation-learning-from#ran","syntology_url":"https://syntology.ai/paper/2009.08319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2009.08319"}},"official":{"repos":["astooke/rlpyt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flightmare-a-flexible-quadrotor-simulator","slug":"flightmare-a-flexible-quadrotor-simulator","title":"Flightmare: A Flexible Quadrotor Simulator","date":"2020-09-01","arxiv_id":"2009.00563","repositories_listed":3,"syntology":null},{"url":"/paper/monte-carlo-tree-search-as-regularized-policy","slug":"monte-carlo-tree-search-as-regularized-policy","title":"Monte-Carlo Tree Search as Regularized Policy Optimization","date":"2020-07-24","arxiv_id":"2007.12509","repositories_listed":3,"syntology":null},{"url":"/paper/uav-path-planning-for-wireless-data","slug":"uav-path-planning-for-wireless-data","title":"UAV Path Planning for Wireless Data Harvesting: A Deep Reinforcement Learning Approach","date":"2020-07-01","arxiv_id":"2007.00544","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-real","slug":"deep-reinforcement-learning-for-real","title":"Deep Reinforcement learning for real autonomous mobile robot navigation in indoor environments","date":"2020-05-28","arxiv_id":"2005.13857","repositories_listed":3,"syntology":null},{"url":"/paper/implementation-matters-in-deep-policy","slug":"implementation-matters-in-deep-policy","title":"Implementation Matters in Deep Policy Gradients: A Case Study on PPO and TRPO","date":"2020-05-25","arxiv_id":"2005.12729","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/implementation-matters-in-deep-policy#ran","syntology_url":"https://syntology.ai/paper/2005.12729","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2005.12729"}},"official":{"repos":["MadryLab/implementation-matters"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/offline-reinforcement-learning-tutorial","slug":"offline-reinforcement-learning-tutorial","title":"Offline Reinforcement Learning: Tutorial, Review, and Perspectives on Open Problems","date":"2020-05-04","arxiv_id":"2005.01643","repositories_listed":3,"syntology":null},{"url":"/paper/ultrasound-guided-robotic-navigation-with","slug":"ultrasound-guided-robotic-navigation-with","title":"Ultrasound-Guided Robotic Navigation with Deep Reinforcement Learning","date":"2020-03-30","arxiv_id":"2003.13321","repositories_listed":3,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ultrasound-guided-robotic-navigation-with#ran","syntology_url":"https://syntology.ai/paper/2003.13321","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.13321"}},"official":{"repos":["hhase/spinal-navigation-rl","hhase/sacrum_data-set"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/sample-efficient-reinforcement-learning","slug":"sample-efficient-reinforcement-learning","title":"Sample Efficient Reinforcement Learning through Learning from Demonstrations in Minecraft","date":"2020-03-12","arxiv_id":"2003.06066","repositories_listed":3,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/sample-efficient-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2003.06066","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2003.06066"}},"official":null}},{"url":"/paper/a-deep-reinforcement-learning-algorithm-using","slug":"a-deep-reinforcement-learning-algorithm-using","title":"A Deep Reinforcement Learning Algorithm Using Dynamic Attention Model for Vehicle Routing Problems","date":"2020-02-09","arxiv_id":"2002.03282","repositories_listed":3,"syntology":null},{"url":"/paper/efficient-object-detection-in-large-images","slug":"efficient-object-detection-in-large-images","title":"Efficient Object Detection in Large Images using Deep Reinforcement Learning","date":"2019-12-09","arxiv_id":"1912.03966","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":2,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-object-detection-in-large-images#ran","syntology_url":"https://syntology.ai/paper/1912.03966","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1912.03966"}},"official":{"repos":["uzkent/EfficientObjectDetection"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/collision-avoidance-in-pedestrian-rich","slug":"collision-avoidance-in-pedestrian-rich","title":"Collision Avoidance in Pedestrian-Rich Environments with Deep Reinforcement Learning","date":"2019-10-24","arxiv_id":"1910.11689","repositories_listed":3,"syntology":null},{"url":"/paper/policies-modulating-trajectory-generators","slug":"policies-modulating-trajectory-generators","title":"Policies Modulating Trajectory Generators","date":"2019-10-07","arxiv_id":"1910.02812","repositories_listed":3,"syntology":null},{"url":"/paper/c-3po-cyclic-three-phase-optimization-for","slug":"c-3po-cyclic-three-phase-optimization-for","title":"C-3PO: Cyclic-Three-Phase Optimization for Human-Robot Motion Retargeting based on Reinforcement Learning","date":"2019-09-25","arxiv_id":"1909.11303","repositories_listed":3,"syntology":null},{"url":"/paper/approximating-two-value-functions-instead-of","slug":"approximating-two-value-functions-instead-of","title":"Approximating two value functions instead of one: towards characterizing a new family of Deep Reinforcement Learning algorithms","date":"2019-09-01","arxiv_id":"1909.01779","repositories_listed":3,"syntology":null},{"url":"/paper/boosting-soft-actor-critic-emphasizing-recent","slug":"boosting-soft-actor-critic-emphasizing-recent","title":"Boosting Soft Actor-Critic: Emphasizing Recent Experience without Forgetting the Past","date":"2019-06-10","arxiv_id":"1906.04009","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/boosting-soft-actor-critic-emphasizing-recent#ran","syntology_url":"https://syntology.ai/paper/1906.04009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1906.04009"}},"official":null}},{"url":"/paper/extrapolating-beyond-suboptimal","slug":"extrapolating-beyond-suboptimal","title":"Extrapolating Beyond Suboptimal Demonstrations via Inverse Reinforcement Learning from Observations","date":"2019-04-12","arxiv_id":"1904.06387","repositories_listed":3,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/extrapolating-beyond-suboptimal#ran","syntology_url":"https://syntology.ai/paper/1904.06387","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1904.06387"}},"official":{"repos":["hiwonjoon/ICML2019-TREX"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/holist-an-environment-for-machine-learning-of","slug":"holist-an-environment-for-machine-learning-of","title":"HOList: An Environment for Machine Learning of Higher-Order Theorem Proving","date":"2019-04-05","arxiv_id":"1904.03241","repositories_listed":3,"syntology":null},{"url":"/paper/improved-robustness-of-reinforcement-learning","slug":"improved-robustness-of-reinforcement-learning","title":"Improved robustness of reinforcement learning policies upon conversion to spiking neuronal network platforms applied to ATARI games","date":"2019-03-26","arxiv_id":"1903.11012","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-imbalanced","slug":"deep-reinforcement-learning-for-imbalanced","title":"Deep Reinforcement Learning for Imbalanced Classification","date":"2019-01-05","arxiv_id":"1901.01379","repositories_listed":3,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-for-imbalanced#ran","syntology_url":"https://syntology.ai/paper/1901.01379","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1901.01379"}},"official":{"repos":["linenus/DRL-For-imbalanced-Classification"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/social-influence-as-intrinsic-motivation-for","slug":"social-influence-as-intrinsic-motivation-for","title":"Social Influence as Intrinsic Motivation for Multi-Agent Deep Reinforcement Learning","date":"2018-10-19","arxiv_id":"1810.08647","repositories_listed":3,"syntology":null},{"url":"/paper/deep-quality-value-dqv-learning","slug":"deep-quality-value-dqv-learning","title":"Deep Quality-Value (DQV) Learning","date":"2018-09-30","arxiv_id":"1810.00368","repositories_listed":3,"syntology":null},{"url":"/paper/dynamic-weights-in-multi-objective-deep","slug":"dynamic-weights-in-multi-objective-deep","title":"Dynamic Weights in Multi-Objective Deep Reinforcement Learning","date":"2018-09-20","arxiv_id":"1809.07803","repositories_listed":3,"syntology":null},{"url":"/paper/surprising-negative-results-for-generative","slug":"surprising-negative-results-for-generative","title":"Surprising Negative Results for Generative Adversarial Tree Search","date":"2018-06-15","arxiv_id":"1806.05780","repositories_listed":3,"syntology":null},{"url":"/paper/maximum-a-posteriori-policy-optimisation","slug":"maximum-a-posteriori-policy-optimisation","title":"Maximum a Posteriori Policy Optimisation","date":"2018-06-14","arxiv_id":"1806.06920","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-sequence-to","slug":"deep-reinforcement-learning-for-sequence-to","title":"Deep Reinforcement Learning For Sequence to Sequence Models","date":"2018-05-24","arxiv_id":"1805.09461","repositories_listed":3,"syntology":{"n":17,"n_ran":16,"n_constructed":0,"n_ran_checked":15,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":15,"n_pointer_only":2,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 0 honoured, 0 violated, 15 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/deep-reinforcement-learning-for-sequence-to#ran","syntology_url":"https://syntology.ai/paper/1805.09461","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1805.09461"}},"official":{"repos":["yaserkl/RLSeq2Seq"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/deep-reinforcement-learning-for-traffic-light","slug":"deep-reinforcement-learning-for-traffic-light","title":"Deep Reinforcement Learning for Traffic Light Control in Vehicular Networks","date":"2018-03-29","arxiv_id":"1803.11115","repositories_listed":3,"syntology":null},{"url":"/paper/jointly-learning-to-construct-and-control","slug":"jointly-learning-to-construct-and-control","title":"Jointly Learning to Construct and Control Agents using Deep Reinforcement Learning","date":"2018-01-04","arxiv_id":"1801.01432","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/jointly-learning-to-construct-and-control#ran","syntology_url":"https://syntology.ai/paper/1801.01432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1801.01432"}},"official":null}},{"url":"/paper/visualizing-and-understanding-atari-agents","slug":"visualizing-and-understanding-atari-agents","title":"Visualizing and Understanding Atari Agents","date":"2017-10-31","arxiv_id":"1711.00138","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/visualizing-and-understanding-atari-agents#ran","syntology_url":"https://syntology.ai/paper/1711.00138","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1711.00138"}},"official":{"repos":["greydanus/visualize_atari"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/molecular-de-novo-design-through-deep","slug":"molecular-de-novo-design-through-deep","title":"Molecular De Novo Design through Deep Reinforcement Learning","date":"2017-04-25","arxiv_id":"1704.07555","repositories_listed":3,"syntology":null},{"url":"/paper/virtual-to-real-deep-reinforcement-learning","slug":"virtual-to-real-deep-reinforcement-learning","title":"Virtual-to-real Deep Reinforcement Learning: Continuous Control of Mobile Robots for Mapless Navigation","date":"2017-03-01","arxiv_id":"1703.00420","repositories_listed":3,"syntology":null},{"url":"/paper/cryptocurrency-portfolio-management-with-deep","slug":"cryptocurrency-portfolio-management-with-deep","title":"Cryptocurrency Portfolio Management with Deep Reinforcement Learning","date":"2016-12-05","arxiv_id":"1612.01277","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cryptocurrency-portfolio-management-with-deep#ran","syntology_url":"https://syntology.ai/paper/1612.01277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1612.01277"}},"official":{"repos":["ZhengyaoJiang/PGPortfolio"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/reinforcement-learning-with-unsupervised","slug":"reinforcement-learning-with-unsupervised","title":"Reinforcement Learning with Unsupervised Auxiliary Tasks","date":"2016-11-16","arxiv_id":"1611.05397","repositories_listed":3,"syntology":null},{"url":"/paper/exploration-a-study-of-count-based","slug":"exploration-a-study-of-count-based","title":"#Exploration: A Study of Count-Based Exploration for Deep Reinforcement Learning","date":"2016-11-15","arxiv_id":"1611.04717","repositories_listed":3,"syntology":null},{"url":"/paper/model-free-episodic-control","slug":"model-free-episodic-control","title":"Model-Free Episodic Control","date":"2016-06-14","arxiv_id":"1606.04460","repositories_listed":3,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/model-free-episodic-control#ran","syntology_url":"https://syntology.ai/paper/1606.04460","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1606.04460"}},"official":null}},{"url":"/paper/actor-mimic-deep-multitask-and-transfer","slug":"actor-mimic-deep-multitask-and-transfer","title":"Actor-Mimic: Deep Multitask and Transfer Reinforcement Learning","date":"2015-11-19","arxiv_id":"1511.06342","repositories_listed":3,"syntology":null},{"url":"/paper/active-object-localization-with-deep","slug":"active-object-localization-with-deep","title":"Active Object Localization with Deep Reinforcement Learning","date":"2015-11-18","arxiv_id":"1511.06015","repositories_listed":3,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-a-natural","slug":"deep-reinforcement-learning-with-a-natural","title":"Deep Reinforcement Learning with a Natural Language Action Space","date":"2015-11-14","arxiv_id":"1511.04636","repositories_listed":3,"syntology":null},{"url":"/paper/giraffe-using-deep-reinforcement-learning-to","slug":"giraffe-using-deep-reinforcement-learning-to","title":"Giraffe: Using Deep Reinforcement Learning to Play Chess","date":"2015-09-04","arxiv_id":"1509.01549","repositories_listed":3,"syntology":null},{"url":"/paper/massively-parallel-methods-for-deep","slug":"massively-parallel-methods-for-deep","title":"Massively Parallel Methods for Deep Reinforcement Learning","date":"2015-07-15","arxiv_id":"1507.04296","repositories_listed":3,"syntology":null},{"url":"/paper/language-understanding-for-text-based-games","slug":"language-understanding-for-text-based-games","title":"Language Understanding for Text-based Games Using Deep Reinforcement Learning","date":"2015-06-30","arxiv_id":"1506.08941","repositories_listed":3,"syntology":null},{"url":"/paper/charms-cognitive-hierarchical-agent-with","slug":"charms-cognitive-hierarchical-agent-with","title":"CHARMS: A Cognitive Hierarchical Agent for Reasoning and Motion Stylization in Autonomous Driving","date":"2025-04-03","arxiv_id":"2504.02450","repositories_listed":2,"syntology":null},{"url":"/paper/towards-optimal-adversarial-robust","slug":"towards-optimal-adversarial-robust","title":"Towards Optimal Adversarial Robust Reinforcement Learning with Infinity Measurement Error","date":"2025-02-23","arxiv_id":"2502.16734","repositories_listed":2,"syntology":null},{"url":"/paper/reevaluating-policy-gradient-methods-for","slug":"reevaluating-policy-gradient-methods-for","title":"Reevaluating Policy Gradient Methods for Imperfect-Information Games","date":"2025-02-13","arxiv_id":"2502.08938","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reevaluating-policy-gradient-methods-for#ran","syntology_url":"https://syntology.ai/paper/2502.08938","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.08938"}},"official":{"repos":["gabrfarina/exp-a-spiel","nathanlct/iig-rl-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/adopt-modified-adam-can-converge-with-any-b-2","slug":"adopt-modified-adam-can-converge-with-any-b-2","title":"ADOPT: Modified Adam Can Converge with Any $β_2$ with the Optimal Rate","date":"2024-11-05","arxiv_id":"2411.02853","repositories_listed":2,"syntology":{"n":23,"n_ran":14,"n_constructed":0,"n_ran_checked":11,"n_instrument":3,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":4,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 3 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/adopt-modified-adam-can-converge-with-any-b-2#ran","syntology_url":"https://syntology.ai/paper/2411.02853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.02853"}},"official":{"repos":["ishohei220/adopt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/sparsifying-parametric-models-with-l0","slug":"sparsifying-parametric-models-with-l0","title":"Sparsifying Parametric Models with L0 Regularization","date":"2024-09-05","arxiv_id":"2409.03489","repositories_listed":2,"syntology":null},{"url":"/paper/rl-adn-a-high-performance-deep-reinforcement","slug":"rl-adn-a-high-performance-deep-reinforcement","title":"RL-ADN: A High-Performance Deep Reinforcement Learning Environment for Optimal Energy Storage Systems Dispatch in Active Distribution Networks","date":"2024-08-07","arxiv_id":"2408.03685","repositories_listed":2,"syntology":null},{"url":"/paper/unveiling-the-decision-making-process-in","slug":"unveiling-the-decision-making-process-in","title":"Unveiling the Decision-Making Process in Reinforcement Learning with Genetic Programming","date":"2024-07-20","arxiv_id":"2407.14714","repositories_listed":2,"syntology":null},{"url":"/paper/on-robust-reinforcement-learning-with","slug":"on-robust-reinforcement-learning-with","title":"On Robust Reinforcement Learning with Lipschitz-Bounded Policy Networks","date":"2024-05-19","arxiv_id":"2405.11432","repositories_listed":2,"syntology":null},{"url":"/paper/the-curse-of-diversity-in-ensemble-based","slug":"the-curse-of-diversity-in-ensemble-based","title":"The Curse of Diversity in Ensemble-Based Exploration","date":"2024-05-07","arxiv_id":"2405.04342","repositories_listed":2,"syntology":{"n":16,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":13,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/the-curse-of-diversity-in-ensemble-based#ran","syntology_url":"https://syntology.ai/paper/2405.04342","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04342"}},"official":{"repos":["zhixuan-lin/ensemble-rl-continuous","zhixuan-lin/ensemble-rl-discrete"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":13,"ran_from_kinds":["official"]}}},{"url":"/paper/peersimgym-an-environment-for-solving-the","slug":"peersimgym-an-environment-for-solving-the","title":"PeersimGym: An Environment for Solving the Task Offloading Problem with Reinforcement Learning","date":"2024-03-26","arxiv_id":"2403.17637","repositories_listed":2,"syntology":null},{"url":"/paper/parametric-pde-control-with-deep","slug":"parametric-pde-control-with-deep","title":"Parametric PDE Control with Deep Reinforcement Learning and Differentiable L0-Sparse Polynomial Policies","date":"2024-03-22","arxiv_id":"2403.15267","repositories_listed":2,"syntology":null},{"url":"/paper/combinatorial-client-master-multiagent-deep","slug":"combinatorial-client-master-multiagent-deep","title":"Combinatorial Client-Master Multiagent Deep Reinforcement Learning for Task Offloading in Mobile Edge Computing","date":"2024-02-18","arxiv_id":"2402.11653","repositories_listed":2,"syntology":null},{"url":"/paper/reinforcement-learning-as-a-catalyst-for","slug":"reinforcement-learning-as-a-catalyst-for","title":"FedAA: A Reinforcement Learning Perspective on Adaptive Aggregation for Fair and Robust Federated Learning","date":"2024-02-08","arxiv_id":"2402.05541","repositories_listed":2,"syntology":null},{"url":"/paper/towards-optimal-adversarial-robust-q-learning","slug":"towards-optimal-adversarial-robust-q-learning","title":"Towards Optimal Adversarial Robust Q-learning with Bellman Infinity-error","date":"2024-02-03","arxiv_id":"2402.02165","repositories_listed":2,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/towards-optimal-adversarial-robust-q-learning#ran","syntology_url":"https://syntology.ai/paper/2402.02165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.02165"}},"official":{"repos":["leoranlmia/CAR-DQN"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/exploration-and-anti-exploration-with","slug":"exploration-and-anti-exploration-with","title":"Exploration and Anti-Exploration with Distributional Random Network Distillation","date":"2024-01-18","arxiv_id":"2401.09750","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploration-and-anti-exploration-with#ran","syntology_url":"https://syntology.ai/paper/2401.09750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.09750"}},"official":{"repos":["yk7333/drnd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autonomous-driving-using-residual-sensor","slug":"autonomous-driving-using-residual-sensor","title":"Autonomous Driving using Residual Sensor Fusion and Deep Reinforcement Learning","date":"2023-12-27","arxiv_id":"2312.16620","repositories_listed":2,"syntology":null},{"url":"/paper/pdit-interleaving-perception-and-decision","slug":"pdit-interleaving-perception-and-decision","title":"PDiT: Interleaving Perception and Decision-making Transformers for Deep Reinforcement Learning","date":"2023-12-26","arxiv_id":"2312.15863","repositories_listed":2,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-image-to","slug":"deep-reinforcement-learning-for-image-to","title":"RL-I2IT: Image-to-Image Translation with Deep Reinforcement Learning","date":"2023-09-24","arxiv_id":"2309.13672","repositories_listed":2,"syntology":null},{"url":"/paper/ixdrl-a-novel-explainable-deep-reinforcement","slug":"ixdrl-a-novel-explainable-deep-reinforcement","title":"IxDRL: A Novel Explainable Deep Reinforcement Learning Toolkit based on Analyses of Interestingness","date":"2023-07-18","arxiv_id":"2307.08933","repositories_listed":2,"syntology":null},{"url":"/paper/pid-inspired-inductive-biases-for-deep-1","slug":"pid-inspired-inductive-biases-for-deep-1","title":"PID-Inspired Inductive Biases for Deep Reinforcement Learning in Partially Observable Control Tasks","date":"2023-07-12","arxiv_id":"2307.05891","repositories_listed":2,"syntology":{"n":9,"n_ran":8,"n_constructed":1,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 1 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pid-inspired-inductive-biases-for-deep-1#ran","syntology_url":"https://syntology.ai/paper/2307.05891","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.05891"}},"official":null}},{"url":"/paper/energy-optimization-for-hvac-systems-in-multi","slug":"energy-optimization-for-hvac-systems-in-multi","title":"Energy Optimization for HVAC Systems in Multi-VAV Open Offices: A Deep Reinforcement Learning Approach","date":"2023-06-23","arxiv_id":"2306.13333","repositories_listed":2,"syntology":null},{"url":"/paper/for-sale-state-action-representation-learning-1","slug":"for-sale-state-action-representation-learning-1","title":"For SALE: State-Action Representation Learning for Deep Reinforcement Learning","date":"2023-06-04","arxiv_id":"2306.02451","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/for-sale-state-action-representation-learning-1#ran","syntology_url":"https://syntology.ai/paper/2306.02451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02451"}},"official":{"repos":["sfujim/td7"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/nasimemu-network-attack-simulator-emulator","slug":"nasimemu-network-attack-simulator-emulator","title":"NASimEmu: Network Attack Simulator & Emulator for Training Agents Generalizing to Novel Scenarios","date":"2023-05-26","arxiv_id":"2305.17246","repositories_listed":2,"syntology":null},{"url":"/paper/pointerformer-deep-reinforced-multi-pointer","slug":"pointerformer-deep-reinforced-multi-pointer","title":"Pointerformer: Deep Reinforced Multi-Pointer Transformer for the Traveling Salesman Problem","date":"2023-04-19","arxiv_id":"2304.09407","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pointerformer-deep-reinforced-multi-pointer#ran","syntology_url":"https://syntology.ai/paper/2304.09407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.09407"}},"official":{"repos":["Learning4Optimization-HUST/Pointerformer"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["named_in_paper"]}}},{"url":"/paper/ams-drl-learning-multi-pursuit-evasion-for","slug":"ams-drl-learning-multi-pursuit-evasion-for","title":"Learning Multi-Pursuit Evasion for Safe Targeted Navigation of Drones","date":"2023-04-07","arxiv_id":"2304.03443","repositories_listed":2,"syntology":null},{"url":"/paper/targeted-adversarial-attacks-on-deep","slug":"targeted-adversarial-attacks-on-deep","title":"Targeted Adversarial Attacks on Deep Reinforcement Learning Policies via Model Checking","date":"2022-12-10","arxiv_id":"2212.05337","repositories_listed":2,"syntology":null},{"url":"/paper/maskplace-fast-chip-placement-via-reinforced","slug":"maskplace-fast-chip-placement-via-reinforced","title":"MaskPlace: Fast Chip Placement via Reinforced Visual Representation Learning","date":"2022-11-24","arxiv_id":"2211.13382","repositories_listed":2,"syntology":null},{"url":"/paper/protox-explaining-a-reinforcement-learning","slug":"protox-explaining-a-reinforcement-learning","title":"ProtoX: Explaining a Reinforcement Learning Agent via Prototyping","date":"2022-11-06","arxiv_id":"2211.03162","repositories_listed":2,"syntology":null},{"url":"/paper/dextreme-transfer-of-agile-in-hand","slug":"dextreme-transfer-of-agile-in-hand","title":"DeXtreme: Transfer of Agile In-hand Manipulation from Simulation to Reality","date":"2022-10-25","arxiv_id":"2210.13702","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dextreme-transfer-of-agile-in-hand#ran","syntology_url":"https://syntology.ai/paper/2210.13702","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.13702"}},"official":{"repos":["Denys88/rl_games","NVIDIA-Omniverse/IsaacGymEnvs"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-based-joint-2","slug":"deep-reinforcement-learning-based-joint-2","title":"Deep Reinforcement Learning Based Joint Downlink Beamforming and RIS Configuration in RIS-aided MU-MISO Systems Under Hardware Impairments and Imperfect CSI","date":"2022-10-10","arxiv_id":"2211.09702","repositories_listed":2,"syntology":null},{"url":"/paper/discovering-faster-matrix-multiplication","slug":"discovering-faster-matrix-multiplication","title":"Discovering faster matrix multiplication algorithms with reinforcement learning","date":"2022-10-05","arxiv_id":null,"repositories_listed":2,"syntology":null},{"url":"/paper/real-time-reinforcement-learning-for-vision","slug":"real-time-reinforcement-learning-for-vision","title":"Real-Time Reinforcement Learning for Vision-Based Robotics Utilizing Local and Remote Computers","date":"2022-10-05","arxiv_id":"2210.02317","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/real-time-reinforcement-learning-for-vision#ran","syntology_url":"https://syntology.ai/paper/2210.02317","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.02317"}},"official":{"repos":["rlai-lab/relod","rlai-lab/remote-onboard-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-intrinsically-motivated-exploration-in","slug":"deep-intrinsically-motivated-exploration-in","title":"Deep Intrinsically Motivated Exploration in Continuous Control","date":"2022-10-01","arxiv_id":"2210.00293","repositories_listed":2,"syntology":null},{"url":"/paper/programmable-and-customized-intelligence-for","slug":"programmable-and-customized-intelligence-for","title":"Programmable and Customized Intelligence for Traffic Steering in 5G Networks Using Open RAN Architectures","date":"2022-09-28","arxiv_id":"2209.14171","repositories_listed":2,"syntology":null},{"url":"/paper/transformers-are-sample-efficient-world","slug":"transformers-are-sample-efficient-world","title":"Transformers are Sample-Efficient World Models","date":"2022-09-01","arxiv_id":"2209.00588","repositories_listed":2,"syntology":{"n":26,"n_ran":17,"n_constructed":13,"n_ran_checked":16,"n_instrument":1,"n_unverified":9,"n_honours":1,"n_violates":0,"n_no_contract":15,"n_pointer_only":26,"phrase":"17 ran (of which 13 constructed an object rather than computing a result; 16 with no instrument failure: 1 honoured, 0 violated, 15 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/transformers-are-sample-efficient-world#ran","syntology_url":"https://syntology.ai/paper/2209.00588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.00588"}},"official":{"repos":["eloialonso/iris"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":13,"n_ran_no_instrument_failure":16,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/bsac-bayesian-strategy-network-based-soft","slug":"bsac-bayesian-strategy-network-based-soft","title":"Bayesian Soft Actor-Critic: A Directed Acyclic Strategy Graph Based Deep Reinforcement Learning","date":"2022-08-11","arxiv_id":"2208.06033","repositories_listed":2,"syntology":null},{"url":"/paper/automating-dbscan-via-deep-reinforcement","slug":"automating-dbscan-via-deep-reinforcement","title":"Automating DBSCAN via Deep Reinforcement Learning","date":"2022-08-09","arxiv_id":"2208.04537","repositories_listed":2,"syntology":null},{"url":"/paper/a-deep-reinforcement-learning-approach-for-13","slug":"a-deep-reinforcement-learning-approach-for-13","title":"A Deep Reinforcement Learning Approach for Finding Non-Exploitable Strategies in Two-Player Atari Games","date":"2022-07-18","arxiv_id":"2207.08894","repositories_listed":2,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/a-deep-reinforcement-learning-approach-for-13#ran","syntology_url":"https://syntology.ai/paper/2207.08894","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.08894"}},"official":{"repos":["quantumiracle/mars","quantumiracle/nash-dqn"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}}],"record_sha256":"337eae91bb9f2b49c46e9141b2fa9a6e62a8048f35fa73747f327b717bcedc6b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}