{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning/papers/28","list_of":"/task/reinforcement-learning","task":"Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":28,"pages_in_order":132,"rows_per_page":100,"rows":[2701,2800],"of":13178,"counts":{"archive_papers_tagged":13178,"with_a_code_link":4183,"where_syntology_ran_a_sample":1175,"not_listed_spam_title":0,"listed":13178,"listed_where_code_ran":1175,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":988,"every_run_a_failure_of_syntologys_instrument":187,"listed_with_a_run_with_no_instrument_failure":988,"listed_every_run_a_failure_of_syntologys_instrument":187,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning","prev":"/task/reinforcement-learning/papers/27","next":"/task/reinforcement-learning/papers/29","papers":[{"url":"/paper/explainable-reinforcement-learning-for","slug":"explainable-reinforcement-learning-for","title":"Explainable Reinforcement Learning for Longitudinal Control","date":"2021-02-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/longicontrol-a-reinforcement-learning","slug":"longicontrol-a-reinforcement-learning","title":"LongiControl: A Reinforcement Learning Environment for Longitudinal Vehicle Control","date":"2021-02-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-smart","slug":"deep-reinforcement-learning-for-smart","title":"Deep reinforcement learning for smart calibration of radio telescopes","date":"2021-02-05","arxiv_id":"2102.03200","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-with","slug":"multi-agent-reinforcement-learning-with","title":"Multi-Agent Reinforcement Learning with Temporal Logic Specifications","date":"2021-02-01","arxiv_id":"2102.00582","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2102.00582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.00582"}},"official":{"repos":["lrhammond/almanac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/variation-resistant-q-learning-controlling","slug":"variation-resistant-q-learning-controlling","title":"Variation-resistant Q-learning: Controlling and Utilizing Estimation Bias in Reinforcement Learning for Better Performance","date":"2021-02-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/re-hamiltonian-generative-networks","slug":"re-hamiltonian-generative-networks","title":"[Re] Hamiltonian Generative Networks","date":"2021-01-31","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-impact-of-tunable-agents-in","slug":"exploring-the-impact-of-tunable-agents-in","title":"Exploring the Impact of Tunable Agents in Sequential Social Dilemmas","date":"2021-01-28","arxiv_id":"2101.11967","repositories_listed":1,"syntology":null},{"url":"/paper/learning-synthetic-environments-for","slug":"learning-synthetic-environments-for","title":"Learning Synthetic Environments for Reinforcement Learning with Evolution Strategies","date":"2021-01-24","arxiv_id":"2101.09721","repositories_listed":1,"syntology":null},{"url":"/paper/differentiable-trust-region-layers-for-deep-1","slug":"differentiable-trust-region-layers-for-deep-1","title":"Differentiable Trust Region Layers for Deep Reinforcement Learning","date":"2021-01-22","arxiv_id":"2101.09207","repositories_listed":1,"syntology":null},{"url":"/paper/theory-of-mind-for-deep-reinforcement","slug":"theory-of-mind-for-deep-reinforcement","title":"Theory of Mind for Deep Reinforcement Learning in Hanabi","date":"2021-01-22","arxiv_id":"2101.09328","repositories_listed":1,"syntology":null},{"url":"/paper/mt5b3-a-framework-for-building","slug":"mt5b3-a-framework-for-building","title":"mt5se: An Open Source Framework for Building Autonomous Trading Robots","date":"2021-01-20","arxiv_id":"2101.08169","repositories_listed":1,"syntology":null},{"url":"/paper/updet-universal-multi-agent-reinforcement","slug":"updet-universal-multi-agent-reinforcement","title":"UPDeT: Universal Multi-agent Reinforcement Learning via Policy Decoupling with Transformers","date":"2021-01-20","arxiv_id":"2101.08001","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-producing","slug":"deep-reinforcement-learning-for-producing","title":"Deep Reinforcement Learning for Producing Furniture Layout in Indoor Scenes","date":"2021-01-19","arxiv_id":"2101.07462","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-active-high","slug":"deep-reinforcement-learning-for-active-high","title":"Deep Reinforcement Learning for Active High Frequency Trading","date":"2021-01-18","arxiv_id":"2101.07107","repositories_listed":1,"syntology":null},{"url":"/paper/interpretable-policy-specification-and","slug":"interpretable-policy-specification-and","title":"Natural Language Specification of Reinforcement Learning Policies through Differentiable Decision Trees","date":"2021-01-18","arxiv_id":"2101.07140","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-by-1","slug":"hierarchical-reinforcement-learning-by-1","title":"Hierarchical Reinforcement Learning By Discovering Intrinsic Options","date":"2021-01-16","arxiv_id":"2101.06521","repositories_listed":1,"syntology":null},{"url":"/paper/controlling-the-risk-of-conversational-search","slug":"controlling-the-risk-of-conversational-search","title":"Controlling the Risk of Conversational Search via Reinforcement Learning","date":"2021-01-15","arxiv_id":"2101.06327","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-behavioral-similarity-embeddings-1","slug":"contrastive-behavioral-similarity-embeddings-1","title":"Contrastive Behavioral Similarity Embeddings for Generalization in Reinforcement Learning","date":"2021-01-13","arxiv_id":"2101.05265","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-soccer-player-from-live-camera-to","slug":"evaluating-soccer-player-from-live-camera-to","title":"Evaluating Soccer Player: from Live Camera to Deep Reinforcement Learning","date":"2021-01-13","arxiv_id":"2101.05388","repositories_listed":1,"syntology":null},{"url":"/paper/memory-augmented-reinforcement-learning-for","slug":"memory-augmented-reinforcement-learning-for","title":"Memory-Augmented Reinforcement Learning for Image-Goal Navigation","date":"2021-01-13","arxiv_id":"2101.05181","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-unlikelihood-training-improving","slug":"implicit-unlikelihood-training-improving","title":"Implicit Unlikelihood Training: Improving Neural Text Generation with Reinforcement Learning","date":"2021-01-11","arxiv_id":"2101.04229","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-function","slug":"deep-reinforcement-learning-with-function","title":"Deep Reinforcement Learning with Function Properties in Mean Reversion Strategies","date":"2021-01-09","arxiv_id":"2101.03418","repositories_listed":1,"syntology":null},{"url":"/paper/simulating-sql-injection-vulnerability","slug":"simulating-sql-injection-vulnerability","title":"Simulating SQL Injection Vulnerability Exploitation Using Q-Learning Reinforcement Learning Agents","date":"2021-01-08","arxiv_id":"2101.03118","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-collective","slug":"reinforcement-learning-based-collective","title":"Reinforcement Learning based Collective Entity Alignment with Adaptive Features","date":"2021-01-05","arxiv_id":"2101.01353","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-domain-adaptation-for","slug":"cross-modal-domain-adaptation-for","title":"Cross-Modal Domain Adaptation for Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/faults-in-deep-reinforcement-learning","slug":"faults-in-deep-reinforcement-learning","title":"Faults in Deep Reinforcement Learning Programs: A Taxonomy and A Detection Approach","date":"2021-01-01","arxiv_id":"2101.00135","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-meta-reinforcement-learning-for","slug":"hierarchical-meta-reinforcement-learning-for","title":"Hierarchical Meta Reinforcement Learning for Multi-Task Environments","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/structure-and-randomness-in-planning-and","slug":"structure-and-randomness-in-planning-and","title":"Structure and randomness in planning and reinforcement learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-task-clustering-for-multi-task","slug":"unsupervised-task-clustering-for-multi-task","title":"Unsupervised Task Clustering for Multi-Task Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-7","slug":"multi-agent-reinforcement-learning-for-7","title":"Multi-Agent Reinforcement Learning for Unmanned Aerial Vehicle Coordination by Multi-Critic Policy Gradient Optimization","date":"2020-12-31","arxiv_id":"2012.15472","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-visual-planning-with-self-1","slug":"model-based-visual-planning-with-self-1","title":"Model-Based Visual Planning with Self-Supervised Functional Distances","date":"2020-12-30","arxiv_id":"2012.15373","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-portfolio","slug":"deep-reinforcement-learning-for-portfolio","title":"Deep Reinforcement Learning for Long-Short Portfolio Optimization","date":"2020-12-26","arxiv_id":"2012.13773","repositories_listed":1,"syntology":null},{"url":"/paper/qvmix-and-qvmix-max-extending-the-deep","slug":"qvmix-and-qvmix-max-extending-the-deep","title":"QVMix and QVMix-Max: Extending the Deep Quality-Value Family of Algorithms to Cooperative Multi-Agent Reinforcement Learning","date":"2020-12-22","arxiv_id":"2012.12062","repositories_listed":1,"syntology":null},{"url":"/paper/2012-11662","slug":"2012-11662","title":"Explicitly Encouraging Low Fractional Dimensional Trajectories Via Reinforcement Learning","date":"2020-12-21","arxiv_id":"2012.11662","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-from-images","slug":"offline-reinforcement-learning-from-images","title":"Offline Reinforcement Learning from Images with Latent Space Models","date":"2020-12-21","arxiv_id":"2012.11547","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-joint-1","slug":"deep-reinforcement-learning-for-joint-1","title":"Deep Reinforcement Learning for Joint Spectrum and Power Allocation in Cellular Networks","date":"2020-12-19","arxiv_id":"2012.10682","repositories_listed":1,"syntology":null},{"url":"/paper/high-throughput-synchronous-deep-rl-1","slug":"high-throughput-synchronous-deep-rl-1","title":"High-Throughput Synchronous Deep RL","date":"2020-12-17","arxiv_id":"2012.09849","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/high-throughput-synchronous-deep-rl-1#ran","syntology_url":"https://syntology.ai/paper/2012.09849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.09849"}},"official":{"repos":["IouJenLiu/HTS-RL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-the-efficient-neural-architecture","slug":"improving-the-efficient-neural-architecture","title":"Improving the Efficient Neural Architecture Search via Rewarding Modifications","date":"2020-12-17","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/model-free-and-bayesian-ensembling-model","slug":"model-free-and-bayesian-ensembling-model","title":"Model-free and Bayesian Ensembling Model-based Deep Reinforcement Learning for Particle Accelerator Control Demonstrated on the FERMI FEL","date":"2020-12-17","arxiv_id":"2012.09737","repositories_listed":1,"syntology":null},{"url":"/paper/learning-accurate-long-term-dynamics-for","slug":"learning-accurate-long-term-dynamics-for","title":"Learning Accurate Long-term Dynamics for Model-based Reinforcement Learning","date":"2020-12-16","arxiv_id":"2012.09156","repositories_listed":1,"syntology":null},{"url":"/paper/cloud-database-tuning-with-reinforcement","slug":"cloud-database-tuning-with-reinforcement","title":"Cloud Database Tuning with Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/evolutionary-learning-of-interpretable","slug":"evolutionary-learning-of-interpretable","title":"Evolutionary learning of interpretable decision trees","date":"2020-12-14","arxiv_id":"2012.07723","repositories_listed":1,"syntology":null},{"url":"/paper/increasing-data-efficiency-of-driving-agent","slug":"increasing-data-efficiency-of-driving-agent","title":"Increasing Data Efficiency of Driving Agent By World Model","date":"2020-12-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-contact-rich-tasks","slug":"reinforcement-learning-for-contact-rich-tasks","title":"Reinforcement Learning for Contact-Rich Tasks: Robotic Peg Insertion Strategies","date":"2020-12-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/super-reinforcement-bros-playing-super-mario","slug":"super-reinforcement-bros-playing-super-mario","title":"Super Reinforcement Bros: Playing Super Mario Bros with Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/an-efficient-asynchronous-method-for-1","slug":"an-efficient-asynchronous-method-for-1","title":"An Efficient Asynchronous Method for Integrating Evolutionary and Gradient-based Policy Search","date":"2020-12-10","arxiv_id":"2012.05417","repositories_listed":1,"syntology":null},{"url":"/paper/combining-reinforcement-learning-with-lin","slug":"combining-reinforcement-learning-with-lin","title":"Combining Reinforcement Learning with Lin-Kernighan-Helsgaun Algorithm for the Traveling Salesman Problem","date":"2020-12-08","arxiv_id":"2012.04461","repositories_listed":1,"syntology":null},{"url":"/paper/navrep-unsupervised-representations-for","slug":"navrep-unsupervised-representations-for","title":"NavRep: Unsupervised Representations for Reinforcement Learning of Robot Navigation in Dynamic Human Environments","date":"2020-12-08","arxiv_id":"2012.04406","repositories_listed":1,"syntology":null},{"url":"/paper/gaea-graph-augmentation-for-equitable-access","slug":"gaea-graph-augmentation-for-equitable-access","title":"GAEA: Graph Augmentation for Equitable Access via Reinforcement Learning","date":"2020-12-07","arxiv_id":"2012.03900","repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-maximum-entropy-inverse","slug":"revisiting-maximum-entropy-inverse","title":"Revisiting Maximum Entropy Inverse Reinforcement Learning: New Perspectives and Algorithms","date":"2020-12-01","arxiv_id":"2012.00889","repositories_listed":1,"syntology":null},{"url":"/paper/rl-unplugged-a-collection-of-benchmarks-for","slug":"rl-unplugged-a-collection-of-benchmarks-for","title":"RL Unplugged: A Collection of Benchmarks for Offline Reinforcement Learning","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-the-neural-architecture-of","slug":"optimizing-the-neural-architecture-of","title":"Optimizing the Neural Architecture of Reinforcement Learning Agents","date":"2020-11-30","arxiv_id":"2011.14632","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-visual-reinforcement-learning-1","slug":"self-supervised-visual-reinforcement-learning-1","title":"Self-supervised Visual Reinforcement Learning with Object-centric Representations","date":"2020-11-29","arxiv_id":"2011.14381","repositories_listed":1,"syntology":null},{"url":"/paper/an-end-to-end-deep-reinforcement-learning","slug":"an-end-to-end-deep-reinforcement-learning","title":"An End-to-end Deep Reinforcement Learning Approach for the Long-term Short-term Planning on the Frenet Space","date":"2020-11-26","arxiv_id":"2011.13098","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-reinforcement-learning-is-a","slug":"distributed-reinforcement-learning-is-a","title":"RLlib Flow: Distributed Reinforcement Learning is a Dataflow Problem","date":"2020-11-25","arxiv_id":"2011.12719","repositories_listed":1,"syntology":null},{"url":"/paper/learning-principle-of-least-action-with","slug":"learning-principle-of-least-action-with","title":"Learning Principle of Least Action with Reinforcement Learning","date":"2020-11-24","arxiv_id":"2011.11891","repositories_listed":1,"syntology":null},{"url":"/paper/evolutionary-planning-in-latent-space","slug":"evolutionary-planning-in-latent-space","title":"Evolutionary Planning in Latent Space","date":"2020-11-23","arxiv_id":"2011.11293","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-exploration-for-model-based-1","slug":"efficient-exploration-for-model-based-1","title":"Model-based Reinforcement Learning for Continuous Control with Posterior Sampling","date":"2020-11-20","arxiv_id":"2012.09613","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/efficient-exploration-for-model-based-1#ran","syntology_url":"https://syntology.ai/paper/2012.09613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.09613"}},"official":{"repos":["yingfan-bot/mbpsrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inverse-constrained-reinforcement-learning","slug":"inverse-constrained-reinforcement-learning","title":"Inverse Constrained Reinforcement Learning","date":"2020-11-19","arxiv_id":"2011.09999","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/inverse-constrained-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2011.09999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.09999"}},"official":{"repos":["shehryar-malik/icrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-clustering-in-particle-physics","slug":"hierarchical-clustering-in-particle-physics","title":"Hierarchical clustering in particle physics through reinforcement learning","date":"2020-11-16","arxiv_id":"2011.08191","repositories_listed":1,"syntology":null},{"url":"/paper/scalable-reinforcement-learning-policies-for","slug":"scalable-reinforcement-learning-policies-for","title":"Scalable Reinforcement Learning Policies for Multi-Agent Control","date":"2020-11-16","arxiv_id":"2011.08055","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scalable-reinforcement-learning-policies-for#ran","syntology_url":"https://syntology.ai/paper/2011.08055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.08055"}},"official":{"repos":["christopher-hsu/scalableMARL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cdt-cascading-decision-trees-for-explainable-1","slug":"cdt-cascading-decision-trees-for-explainable-1","title":"CDT: Cascading Decision Trees for Explainable Reinforcement Learning","date":"2020-11-15","arxiv_id":"2011.07553","repositories_listed":1,"syntology":null},{"url":"/paper/deepmind-lab2d","slug":"deepmind-lab2d","title":"DeepMind Lab2D","date":"2020-11-13","arxiv_id":"2011.07027","repositories_listed":1,"syntology":null},{"url":"/paper/roll-visual-self-supervised-reinforcement","slug":"roll-visual-self-supervised-reinforcement","title":"ROLL: Visual Self-Supervised Reinforcement Learning with Object Reasoning","date":"2020-11-13","arxiv_id":"2011.06777","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-large-scale-fleet-management-on-a","slug":"optimizing-large-scale-fleet-management-on-a","title":"Optimizing Large-Scale Fleet Management on a Road Network using Multi-Agent Deep Reinforcement Learning with Graph Neural Network","date":"2020-11-12","arxiv_id":"2011.06175","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-videos-combining","slug":"reinforcement-learning-with-videos-combining","title":"Reinforcement Learning with Videos: Combining Offline Observations with Interaction","date":"2020-11-12","arxiv_id":"2011.06507","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-experiments-and","slug":"reinforcement-learning-experiments-and","title":"Reinforcement Learning Experiments and Benchmark for Solving Robotic Reaching Tasks","date":"2020-11-11","arxiv_id":"2011.05782","repositories_listed":1,"syntology":null},{"url":"/paper/robust-reinforcement-learning-for-general","slug":"robust-reinforcement-learning-for-general","title":"Reinforcement Learning with Dual-Observation for General Video Game Playing","date":"2020-11-11","arxiv_id":"2011.05622","repositories_listed":1,"syntology":null},{"url":"/paper/f-irl-inverse-reinforcement-learning-via","slug":"f-irl-inverse-reinforcement-learning-via","title":"f-IRL: Inverse Reinforcement Learning via State Marginal Matching","date":"2020-11-09","arxiv_id":"2011.04709","repositories_listed":1,"syntology":null},{"url":"/paper/geometric-deep-reinforcement-learning-for","slug":"geometric-deep-reinforcement-learning-for","title":"Geometric Deep Reinforcement Learning for Dynamic DAG Scheduling","date":"2020-11-09","arxiv_id":"2011.04333","repositories_listed":1,"syntology":null},{"url":"/paper/trajectory-planning-for-autonomous-vehicles","slug":"trajectory-planning-for-autonomous-vehicles","title":"Trajectory Planning for Autonomous Vehicles Using Hierarchical Reinforcement Learning","date":"2020-11-09","arxiv_id":"2011.04752","repositories_listed":1,"syntology":null},{"url":"/paper/drafting-in-collectible-card-games-via","slug":"drafting-in-collectible-card-games-via","title":"Drafting in Collectible Card Games via Reinforcement Learning","date":"2020-11-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-a-decentralized-multi-arm-motion","slug":"learning-a-decentralized-multi-arm-motion","title":"Learning a Decentralized Multi-arm Motion Planner","date":"2020-11-05","arxiv_id":"2011.02608","repositories_listed":1,"syntology":null},{"url":"/paper/realant-an-open-source-low-cost-quadruped-for","slug":"realant-an-open-source-low-cost-quadruped-for","title":"RealAnt: An Open-Source Low-Cost Quadruped for Education and Research in Real-World Reinforcement Learning","date":"2020-11-05","arxiv_id":"2011.03085","repositories_listed":1,"syntology":null},{"url":"/paper/learning-trajectories-for-visual-inertial","slug":"learning-trajectories-for-visual-inertial","title":"Learning Trajectories for Visual-Inertial System Calibration via Model-based Heuristic Deep Reinforcement Learning","date":"2020-11-04","arxiv_id":"2011.02574","repositories_listed":1,"syntology":null},{"url":"/paper/causal-campbell-goodhart-s-law-and","slug":"causal-campbell-goodhart-s-law-and","title":"Causal Campbell-Goodhart's law and Reinforcement Learning","date":"2020-11-02","arxiv_id":"2011.01010","repositories_listed":1,"syntology":null},{"url":"/paper/instance-based-generalization-in","slug":"instance-based-generalization-in","title":"Instance based Generalization in Reinforcement Learning","date":"2020-11-02","arxiv_id":"2011.01089","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":4,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/instance-based-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2011.01089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.01089"}},"official":{"repos":["MartinBertran/InstanceAgnosticPolicyEnsembles"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-reinforcement-learning-for","slug":"multi-agent-reinforcement-learning-for","title":"Multi-Agent Reinforcement Learning for Visibility-based Persistent Monitoring","date":"2020-11-02","arxiv_id":"2011.01129","repositories_listed":1,"syntology":null},{"url":"/paper/self-driving-network-and-service-coordination","slug":"self-driving-network-and-service-coordination","title":"Self-Driving Network and Service Coordination Using Deep Reinforcement Learning","date":"2020-11-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/an-overview-of-multi-agent-reinforcement","slug":"an-overview-of-multi-agent-reinforcement","title":"Game-Theoretic Multiagent Reinforcement Learning","date":"2020-11-01","arxiv_id":"2011.00583","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/an-overview-of-multi-agent-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2011.00583","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.00583"}},"official":null}},{"url":"/paper/guided-dialogue-policy-learning-without","slug":"guided-dialogue-policy-learning-without","title":"Guided Dialogue Policy Learning without Adversarial Learning in the Loop","date":"2020-11-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/personalized-multimorbidity-management-for","slug":"personalized-multimorbidity-management-for","title":"Personalized Multimorbidity Management for Patients with Type 2 Diabetes Using Reinforcement Learning of Electronic Health Records","date":"2020-10-29","arxiv_id":"2011.02287","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-under-parameterization-inhibits-data-1","slug":"implicit-under-parameterization-inhibits-data-1","title":"Implicit Under-Parameterization Inhibits Data-Efficient Deep Reinforcement Learning","date":"2020-10-27","arxiv_id":"2010.14498","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/implicit-under-parameterization-inhibits-data-1#ran","syntology_url":"https://syntology.ai/paper/2010.14498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.14498"}},"official":null}},{"url":"/paper/improving-reinforcement-learning-for-neural","slug":"improving-reinforcement-learning-for-neural","title":"RH-Net: Improving Neural Relation Extraction via Reinforcement Learning and Hierarchical Relational Searching","date":"2020-10-27","arxiv_id":"2010.14255","repositories_listed":1,"syntology":null},{"url":"/paper/learning-financial-asset-specific-trading","slug":"learning-financial-asset-specific-trading","title":"Learning Financial Asset-Specific Trading Rules via Deep Reinforcement Learning","date":"2020-10-27","arxiv_id":"2010.14194","repositories_listed":1,"syntology":null},{"url":"/paper/meld-meta-reinforcement-learning-from-images","slug":"meld-meta-reinforcement-learning-from-images","title":"MELD: Meta-Reinforcement Learning from Images via Latent State Models","date":"2020-10-26","arxiv_id":"2010.13957","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-make-deep-rl-work-in-practice","slug":"how-to-make-deep-rl-work-in-practice","title":"How to Make Deep RL Work in Practice","date":"2020-10-25","arxiv_id":"2010.13083","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-imagination-and-reality-for-model","slug":"bridging-imagination-and-reality-for-model","title":"Bridging Imagination and Reality for Model-Based Deep Reinforcement Learning","date":"2020-10-23","arxiv_id":"2010.12142","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-imagination-and-reality-for-model#ran","syntology_url":"https://syntology.ai/paper/2010.12142","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.12142"}},"official":{"repos":["Mehooz/BIRD_code"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/batch-exploration-with-examples-for-scalable","slug":"batch-exploration-with-examples-for-scalable","title":"Batch Exploration with Examples for Scalable Robotic Reinforcement Learning","date":"2020-10-22","arxiv_id":"2010.11917","repositories_listed":1,"syntology":null},{"url":"/paper/drift-detection-in-episodic-data-detect-when-1","slug":"drift-detection-in-episodic-data-detect-when-1","title":"Detecting Rewards Deterioration in Episodic Reinforcement Learning","date":"2020-10-22","arxiv_id":"2010.11660","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/drift-detection-in-episodic-data-detect-when-1#ran","syntology_url":"https://syntology.ai/paper/2010.11660","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.11660"}},"official":{"repos":["ido90/Rewards-Deterioration-Detection"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-combinatorial","slug":"reinforcement-learning-with-combinatorial","title":"Reinforcement Learning with Combinatorial Actions: An Application to Vehicle Routing","date":"2020-10-22","arxiv_id":"2010.12001","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-optimization-of","slug":"reinforcement-learning-for-optimization-of","title":"Reinforcement Learning for Optimization of COVID-19 Mitigation policies","date":"2020-10-20","arxiv_id":"2010.10560","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforcement-learning-for-optimization-of#ran","syntology_url":"https://syntology.ai/paper/2010.10560","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10560"}},"official":{"repos":["SonyAI/PandemicSimulator"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/every-hidden-unit-maximizing-output-weights","slug":"every-hidden-unit-maximizing-output-weights","title":"Learning by Competition of Self-Interested Reinforcement Learning Agents","date":"2020-10-19","arxiv_id":"2010.09770","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-guided-open-attribute-value","slug":"knowledge-guided-open-attribute-value","title":"Knowledge-guided Open Attribute Value Extraction with Reinforcement Learning","date":"2020-10-19","arxiv_id":"2010.09189","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/knowledge-guided-open-attribute-value#ran","syntology_url":"https://syntology.ai/paper/2010.09189","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09189"}},"official":{"repos":["yeliu0930/Knowledge-guided-Open-Attribute-Value-Extraction-with-Reinforcement-Learning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/model-based-policy-optimization-with","slug":"model-based-policy-optimization-with","title":"Model-based Policy Optimization with Unsupervised Model Adaptation","date":"2020-10-19","arxiv_id":"2010.09546","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":3,"n_honours":2,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/model-based-policy-optimization-with#ran","syntology_url":"https://syntology.ai/paper/2010.09546","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09546"}},"official":{"repos":["RockySJ/ampo"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/a-game-theoretic-analysis-of-networked-system","slug":"a-game-theoretic-analysis-of-networked-system","title":"A game-theoretic analysis of networked system control for common-pool resource management using multi-agent reinforcement learning","date":"2020-10-15","arxiv_id":"2010.07777","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":12,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":1,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/a-game-theoretic-analysis-of-networked-system#ran","syntology_url":"https://syntology.ai/paper/2010.07777","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07777"}},"official":{"repos":["instadeepai/EGTA-NMARL"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/an-alternative-to-backpropagation-in-deep","slug":"an-alternative-to-backpropagation-in-deep","title":"MAP Propagation Algorithm: Faster Learning with a Team of Reinforcement Learning Agents","date":"2020-10-15","arxiv_id":"2010.07893","repositories_listed":1,"syntology":null},{"url":"/paper/human-guided-robot-behavior-learning-a-gan","slug":"human-guided-robot-behavior-learning-a-gan","title":"Human-guided Robot Behavior Learning: A GAN-assisted Preference-based Reinforcement Learning Approach","date":"2020-10-15","arxiv_id":"2010.07467","repositories_listed":1,"syntology":null},{"url":"/paper/masked-contrastive-representation-learning","slug":"masked-contrastive-representation-learning","title":"Masked Contrastive Representation Learning for Reinforcement Learning","date":"2020-10-15","arxiv_id":"2010.07470","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/masked-contrastive-representation-learning#ran","syntology_url":"https://syntology.ai/paper/2010.07470","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07470"}},"official":{"repos":["teslacool/m-curl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-task-deep-reinforcement-learning-with-1","slug":"multi-task-deep-reinforcement-learning-with-1","title":"Knowledge Transfer in Multi-Task Deep Reinforcement Learning for Continuous Control","date":"2020-10-15","arxiv_id":"2010.07494","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-deep-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2010.07494","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07494"}},"official":null}}],"record_sha256":"e1a759006614b0d36d80b9f283cc9284c45017153912badbbcc1c38b1b2cf6fe","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}