{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/35","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":35,"pages_in_order":152,"rows_per_page":100,"rows":[3401,3500],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/34","next":"/task/reinforcement-learning-1/papers/36","papers":[{"url":"/paper/longicontrol-a-reinforcement-learning","slug":"longicontrol-a-reinforcement-learning","title":"LongiControl: A Reinforcement Learning Environment for Longitudinal Vehicle Control","date":"2021-02-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-smart","slug":"deep-reinforcement-learning-for-smart","title":"Deep reinforcement learning for smart calibration of radio telescopes","date":"2021-02-05","arxiv_id":"2102.03200","repositories_listed":1,"syntology":null},{"url":"/paper/gnn-rl-compression-topology-aware-network","slug":"gnn-rl-compression-topology-aware-network","title":"Topology-Aware Network Pruning using Multi-stage Graph Embedding and Reinforcement Learning","date":"2021-02-05","arxiv_id":"2102.03214","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gnn-rl-compression-topology-aware-network#ran","syntology_url":"https://syntology.ai/paper/2102.03214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.03214"}},"official":{"repos":["yusx-swapp/gnn-rl-model-compression"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/alchemy-a-structured-task-distribution-for","slug":"alchemy-a-structured-task-distribution-for","title":"Alchemy: A benchmark and analysis toolkit for meta-reinforcement learning agents","date":"2021-02-04","arxiv_id":"2102.02926","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-with","slug":"multi-agent-reinforcement-learning-with","title":"Multi-Agent Reinforcement Learning with Temporal Logic Specifications","date":"2021-02-01","arxiv_id":"2102.00582","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2102.00582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.00582"}},"official":{"repos":["lrhammond/almanac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/variation-resistant-q-learning-controlling","slug":"variation-resistant-q-learning-controlling","title":"Variation-resistant Q-learning: Controlling and Utilizing Estimation Bias in Reinforcement Learning for Better Performance","date":"2021-02-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/meta-reinforcement-learning-for-reliable","slug":"meta-reinforcement-learning-for-reliable","title":"Meta-Reinforcement Learning for Reliable Communication in THz/VLC Wireless VR Networks","date":"2021-01-29","arxiv_id":"2102.12277","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-impact-of-tunable-agents-in","slug":"exploring-the-impact-of-tunable-agents-in","title":"Exploring the Impact of Tunable Agents in Sequential Social Dilemmas","date":"2021-01-28","arxiv_id":"2101.11967","repositories_listed":1,"syntology":null},{"url":"/paper/data-sharing-games","slug":"data-sharing-games","title":"Data sharing games","date":"2021-01-26","arxiv_id":"2101.10721","repositories_listed":1,"syntology":null},{"url":"/paper/learning-synthetic-environments-for","slug":"learning-synthetic-environments-for","title":"Learning Synthetic Environments for Reinforcement Learning with Evolution Strategies","date":"2021-01-24","arxiv_id":"2101.09721","repositories_listed":1,"syntology":null},{"url":"/paper/bf-a-language-for-general-purpose-neural","slug":"bf-a-language-for-general-purpose-neural","title":"BF++: a language for general-purpose program synthesis","date":"2021-01-23","arxiv_id":"2101.09571","repositories_listed":1,"syntology":null},{"url":"/paper/differentiable-trust-region-layers-for-deep-1","slug":"differentiable-trust-region-layers-for-deep-1","title":"Differentiable Trust Region Layers for Deep Reinforcement Learning","date":"2021-01-22","arxiv_id":"2101.09207","repositories_listed":1,"syntology":null},{"url":"/paper/theory-of-mind-for-deep-reinforcement","slug":"theory-of-mind-for-deep-reinforcement","title":"Theory of Mind for Deep Reinforcement Learning in Hanabi","date":"2021-01-22","arxiv_id":"2101.09328","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-cardiovascular-modelling-with-deep","slug":"unifying-cardiovascular-modelling-with-deep","title":"Unifying Cardiovascular Modelling with Deep Reinforcement Learning for Uncertainty Aware Control of Sepsis Treatment","date":"2021-01-21","arxiv_id":"2101.08477","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-cardiovascular-modelling-with-deep#ran","syntology_url":"https://syntology.ai/paper/2101.08477","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.08477"}},"official":{"repos":["thxsxth/POMDP_RLSepsis"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/mt5b3-a-framework-for-building","slug":"mt5b3-a-framework-for-building","title":"mt5se: An Open Source Framework for Building Autonomous Trading Robots","date":"2021-01-20","arxiv_id":"2101.08169","repositories_listed":1,"syntology":null},{"url":"/paper/updet-universal-multi-agent-reinforcement","slug":"updet-universal-multi-agent-reinforcement","title":"UPDeT: Universal Multi-agent Reinforcement Learning via Policy Decoupling with Transformers","date":"2021-01-20","arxiv_id":"2101.08001","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-producing","slug":"deep-reinforcement-learning-for-producing","title":"Deep Reinforcement Learning for Producing Furniture Layout in Indoor Scenes","date":"2021-01-19","arxiv_id":"2101.07462","repositories_listed":1,"syntology":null},{"url":"/paper/grounding-language-to-entities-and-dynamics","slug":"grounding-language-to-entities-and-dynamics","title":"Grounding Language to Entities and Dynamics for Generalization in Reinforcement Learning","date":"2021-01-19","arxiv_id":"2101.07393","repositories_listed":1,"syntology":null},{"url":"/paper/towards-facilitating-empathic-conversations","slug":"towards-facilitating-empathic-conversations","title":"Towards Facilitating Empathic Conversations in Online Mental Health Support: A Reinforcement Learning Approach","date":"2021-01-19","arxiv_id":"2101.07714","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-active-high","slug":"deep-reinforcement-learning-for-active-high","title":"Deep Reinforcement Learning for Active High Frequency Trading","date":"2021-01-18","arxiv_id":"2101.07107","repositories_listed":1,"syntology":null},{"url":"/paper/hammer-multi-level-coordination-of","slug":"hammer-multi-level-coordination-of","title":"HAMMER: Multi-Level Coordination of Reinforcement Learning Agents via Learned Messaging","date":"2021-01-18","arxiv_id":"2102.00824","repositories_listed":1,"syntology":null},{"url":"/paper/interpretable-policy-specification-and","slug":"interpretable-policy-specification-and","title":"Natural Language Specification of Reinforcement Learning Policies through Differentiable Decision Trees","date":"2021-01-18","arxiv_id":"2101.07140","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-by-1","slug":"hierarchical-reinforcement-learning-by-1","title":"Hierarchical Reinforcement Learning By Discovering Intrinsic Options","date":"2021-01-16","arxiv_id":"2101.06521","repositories_listed":1,"syntology":null},{"url":"/paper/controlling-the-risk-of-conversational-search","slug":"controlling-the-risk-of-conversational-search","title":"Controlling the Risk of Conversational Search via Reinforcement Learning","date":"2021-01-15","arxiv_id":"2101.06327","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-deep-q-learning-with-simulator-for","slug":"continuous-deep-q-learning-with-simulator-for","title":"Continuous Deep Q-Learning with Simulator for Stabilization of Uncertain Discrete-Time Systems","date":"2021-01-13","arxiv_id":"2101.05640","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-behavioral-similarity-embeddings-1","slug":"contrastive-behavioral-similarity-embeddings-1","title":"Contrastive Behavioral Similarity Embeddings for Generalization in Reinforcement Learning","date":"2021-01-13","arxiv_id":"2101.05265","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-soccer-player-from-live-camera-to","slug":"evaluating-soccer-player-from-live-camera-to","title":"Evaluating Soccer Player: from Live Camera to Deep Reinforcement Learning","date":"2021-01-13","arxiv_id":"2101.05388","repositories_listed":1,"syntology":null},{"url":"/paper/memory-augmented-reinforcement-learning-for","slug":"memory-augmented-reinforcement-learning-for","title":"Memory-Augmented Reinforcement Learning for Image-Goal Navigation","date":"2021-01-13","arxiv_id":"2101.05181","repositories_listed":1,"syntology":null},{"url":"/paper/action-priors-for-large-action-spaces-in","slug":"action-priors-for-large-action-spaces-in","title":"Action Priors for Large Action Spaces in Robotics","date":"2021-01-11","arxiv_id":"2101.04178","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-unlikelihood-training-improving","slug":"implicit-unlikelihood-training-improving","title":"Implicit Unlikelihood Training: Improving Neural Text Generation with Reinforcement Learning","date":"2021-01-11","arxiv_id":"2101.04229","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-contrastive-learning-of","slug":"cross-modal-contrastive-learning-of","title":"Cross-Modal Contrastive Learning of Representations for Navigation using Lightweight, Low-Cost Millimeter Wave Radar for Adverse Environmental Conditions","date":"2021-01-10","arxiv_id":"2101.03525","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-function","slug":"deep-reinforcement-learning-with-function","title":"Deep Reinforcement Learning with Function Properties in Mean Reversion Strategies","date":"2021-01-09","arxiv_id":"2101.03418","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-based-encoder","slug":"a-reinforcement-learning-based-encoder","title":"A Reinforcement Learning Based Encoder-Decoder Framework for Learning Stock Trading Rules","date":"2021-01-08","arxiv_id":"2101.03867","repositories_listed":1,"syntology":null},{"url":"/paper/simulating-sql-injection-vulnerability","slug":"simulating-sql-injection-vulnerability","title":"Simulating SQL Injection Vulnerability Exploitation Using Q-Learning Reinforcement Learning Agents","date":"2021-01-08","arxiv_id":"2101.03118","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-collective","slug":"reinforcement-learning-based-collective","title":"Reinforcement Learning based Collective Entity Alignment with Adaptive Features","date":"2021-01-05","arxiv_id":"2101.01353","repositories_listed":1,"syntology":null},{"url":"/paper/a-novel-policy-for-pre-trained-deep","slug":"a-novel-policy-for-pre-trained-deep","title":"A novel policy for pre-trained Deep Reinforcement Learning for Speech Emotion Recognition","date":"2021-01-04","arxiv_id":"2101.00738","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-domain-adaptation-for","slug":"cross-modal-domain-adaptation-for","title":"Cross-Modal Domain Adaptation for Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/faults-in-deep-reinforcement-learning","slug":"faults-in-deep-reinforcement-learning","title":"Faults in Deep Reinforcement Learning Programs: A Taxonomy and A Detection Approach","date":"2021-01-01","arxiv_id":"2101.00135","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-meta-reinforcement-learning-for","slug":"hierarchical-meta-reinforcement-learning-for","title":"Hierarchical Meta Reinforcement Learning for Multi-Task Environments","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-trust-region-learning","slug":"multi-agent-trust-region-learning","title":"Multi-Agent Trust Region Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/structure-and-randomness-in-planning-and","slug":"structure-and-randomness-in-planning-and","title":"Structure and randomness in planning and reinforcement learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/teac-intergrating-trust-region-and-max","slug":"teac-intergrating-trust-region-and-max","title":"TEAC: Intergrating Trust Region and Max Entropy Actor Critic for Continuous Control","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/trust-but-verify-model-based-exploration-in","slug":"trust-but-verify-model-based-exploration-in","title":"Trust, but verify: model-based exploration in sparse reward environments","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-task-clustering-for-multi-task","slug":"unsupervised-task-clustering-for-multi-task","title":"Unsupervised Task Clustering for Multi-Task Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-7","slug":"multi-agent-reinforcement-learning-for-7","title":"Multi-Agent Reinforcement Learning for Unmanned Aerial Vehicle Coordination by Multi-Critic Policy Gradient Optimization","date":"2020-12-31","arxiv_id":"2012.15472","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-visual-planning-with-self-1","slug":"model-based-visual-planning-with-self-1","title":"Model-Based Visual Planning with Self-Supervised Functional Distances","date":"2020-12-30","arxiv_id":"2012.15373","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-portfolio","slug":"deep-reinforcement-learning-for-portfolio","title":"Deep Reinforcement Learning for Long-Short Portfolio Optimization","date":"2020-12-26","arxiv_id":"2012.13773","repositories_listed":1,"syntology":null},{"url":"/paper/popo-pessimistic-offline-policy-optimization","slug":"popo-pessimistic-offline-policy-optimization","title":"POPO: Pessimistic Offline Policy Optimization","date":"2020-12-26","arxiv_id":"2012.13682","repositories_listed":1,"syntology":null},{"url":"/paper/qvmix-and-qvmix-max-extending-the-deep","slug":"qvmix-and-qvmix-max-extending-the-deep","title":"QVMix and QVMix-Max: Extending the Deep Quality-Value Family of Algorithms to Cooperative Multi-Agent Reinforcement Learning","date":"2020-12-22","arxiv_id":"2012.12062","repositories_listed":1,"syntology":null},{"url":"/paper/2012-11662","slug":"2012-11662","title":"Explicitly Encouraging Low Fractional Dimensional Trajectories Via Reinforcement Learning","date":"2020-12-21","arxiv_id":"2012.11662","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-from-images","slug":"offline-reinforcement-learning-from-images","title":"Offline Reinforcement Learning from Images with Latent Space Models","date":"2020-12-21","arxiv_id":"2012.11547","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-joint-1","slug":"deep-reinforcement-learning-for-joint-1","title":"Deep Reinforcement Learning for Joint Spectrum and Power Allocation in Cellular Networks","date":"2020-12-19","arxiv_id":"2012.10682","repositories_listed":1,"syntology":null},{"url":"/paper/generalize-a-small-pre-trained-model-to","slug":"generalize-a-small-pre-trained-model-to","title":"Generalize a Small Pre-trained Model to Arbitrarily Large TSP Instances","date":"2020-12-19","arxiv_id":"2012.10658","repositories_listed":1,"syntology":null},{"url":"/paper/multi-decoder-attention-model-with-embedding","slug":"multi-decoder-attention-model-with-embedding","title":"Multi-Decoder Attention Model with Embedding Glimpse for Solving Vehicle Routing Problems","date":"2020-12-19","arxiv_id":"2012.10638","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-decoder-attention-model-with-embedding#ran","syntology_url":"https://syntology.ai/paper/2012.10638","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.10638"}},"official":{"repos":["liangxinedu/MDAM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/citylearn-standardizing-research-in-multi","slug":"citylearn-standardizing-research-in-multi","title":"CityLearn: Standardizing Research in Multi-Agent Reinforcement Learning for Demand Response and Urban Energy Management","date":"2020-12-18","arxiv_id":"2012.10504","repositories_listed":1,"syntology":null},{"url":"/paper/content-masked-loss-human-like-brush-stroke","slug":"content-masked-loss-human-like-brush-stroke","title":"Content Masked Loss: Human-Like Brush Stroke Planning in a Reinforcement Learning Painting Agent","date":"2020-12-18","arxiv_id":"2012.10043","repositories_listed":1,"syntology":null},{"url":"/paper/high-throughput-synchronous-deep-rl-1","slug":"high-throughput-synchronous-deep-rl-1","title":"High-Throughput Synchronous Deep RL","date":"2020-12-17","arxiv_id":"2012.09849","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/high-throughput-synchronous-deep-rl-1#ran","syntology_url":"https://syntology.ai/paper/2012.09849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.09849"}},"official":{"repos":["IouJenLiu/HTS-RL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-the-efficient-neural-architecture","slug":"improving-the-efficient-neural-architecture","title":"Improving the Efficient Neural Architecture Search via Rewarding Modifications","date":"2020-12-17","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/model-free-and-bayesian-ensembling-model","slug":"model-free-and-bayesian-ensembling-model","title":"Model-free and Bayesian Ensembling Model-based Deep Reinforcement Learning for Particle Accelerator Control Demonstrated on the FERMI FEL","date":"2020-12-17","arxiv_id":"2012.09737","repositories_listed":1,"syntology":null},{"url":"/paper/learning-accurate-long-term-dynamics-for","slug":"learning-accurate-long-term-dynamics-for","title":"Learning Accurate Long-term Dynamics for Model-based Reinforcement Learning","date":"2020-12-16","arxiv_id":"2012.09156","repositories_listed":1,"syntology":null},{"url":"/paper/cloud-database-tuning-with-reinforcement","slug":"cloud-database-tuning-with-reinforcement","title":"Cloud Database Tuning with Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/evolutionary-learning-of-interpretable","slug":"evolutionary-learning-of-interpretable","title":"Evolutionary learning of interpretable decision trees","date":"2020-12-14","arxiv_id":"2012.07723","repositories_listed":1,"syntology":null},{"url":"/paper/increasing-data-efficiency-of-driving-agent","slug":"increasing-data-efficiency-of-driving-agent","title":"Increasing Data Efficiency of Driving Agent By World Model","date":"2020-12-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/policy-gradient-rl-algorithms-as-directed","slug":"policy-gradient-rl-algorithms-as-directed","title":"Policy Gradient RL Algorithms as Directed Acyclic Graphs","date":"2020-12-14","arxiv_id":"2012.07763","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-contact-rich-tasks","slug":"reinforcement-learning-for-contact-rich-tasks","title":"Reinforcement Learning for Contact-Rich Tasks: Robotic Peg Insertion Strategies","date":"2020-12-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sim-to-real-reinforcement-learning-applied-to","slug":"sim-to-real-reinforcement-learning-applied-to","title":"Sim-to-real reinforcement learning applied to end-to-end vehicle control","date":"2020-12-14","arxiv_id":"2012.07461","repositories_listed":1,"syntology":null},{"url":"/paper/super-reinforcement-bros-playing-super-mario","slug":"super-reinforcement-bros-playing-super-mario","title":"Super Reinforcement Bros: Playing Super Mario Bros with Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/an-efficient-asynchronous-method-for-1","slug":"an-efficient-asynchronous-method-for-1","title":"An Efficient Asynchronous Method for Integrating Evolutionary and Gradient-based Policy Search","date":"2020-12-10","arxiv_id":"2012.05417","repositories_listed":1,"syntology":null},{"url":"/paper/combining-reinforcement-learning-with-lin","slug":"combining-reinforcement-learning-with-lin","title":"Combining Reinforcement Learning with Lin-Kernighan-Helsgaun Algorithm for the Traveling Salesman Problem","date":"2020-12-08","arxiv_id":"2012.04461","repositories_listed":1,"syntology":null},{"url":"/paper/models-pixels-and-rewards-evaluating-design","slug":"models-pixels-and-rewards-evaluating-design","title":"Models, Pixels, and Rewards: Evaluating Design Trade-offs in Visual Model-Based Reinforcement Learning","date":"2020-12-08","arxiv_id":"2012.04603","repositories_listed":1,"syntology":null},{"url":"/paper/navrep-unsupervised-representations-for","slug":"navrep-unsupervised-representations-for","title":"NavRep: Unsupervised Representations for Reinforcement Learning of Robot Navigation in Dynamic Human Environments","date":"2020-12-08","arxiv_id":"2012.04406","repositories_listed":1,"syntology":null},{"url":"/paper/resolving-implicit-coordination-in-multi","slug":"resolving-implicit-coordination-in-multi","title":"Resolving Implicit Coordination in Multi-Agent Deep Reinforcement Learning with Deep Q-Networks & Game Theory","date":"2020-12-08","arxiv_id":"2012.09136","repositories_listed":1,"syntology":null},{"url":"/paper/gaea-graph-augmentation-for-equitable-access","slug":"gaea-graph-augmentation-for-equitable-access","title":"GAEA: Graph Augmentation for Equitable Access via Reinforcement Learning","date":"2020-12-07","arxiv_id":"2012.03900","repositories_listed":1,"syntology":null},{"url":"/paper/reset-free-lifelong-learning-with-skill-space-1","slug":"reset-free-lifelong-learning-with-skill-space-1","title":"Reset-Free Lifelong Learning with Skill-Space Planning","date":"2020-12-07","arxiv_id":"2012.03548","repositories_listed":1,"syntology":null},{"url":"/paper/rloc-terrain-aware-legged-locomotion-using","slug":"rloc-terrain-aware-legged-locomotion-using","title":"RLOC: Terrain-Aware Legged Locomotion using Reinforcement Learning and Optimal Control","date":"2020-12-05","arxiv_id":"2012.03094","repositories_listed":1,"syntology":null},{"url":"/paper/acn-sim-an-open-source-simulator-for-data","slug":"acn-sim-an-open-source-simulator-for-data","title":"ACN-Sim: An Open-Source Simulator for Data-Driven Electric Vehicle Charging Research","date":"2020-12-04","arxiv_id":"2012.02809","repositories_listed":1,"syntology":null},{"url":"/paper/can-q-learning-with-graph-networks-learn-a","slug":"can-q-learning-with-graph-networks-learn-a","title":"Can Q-Learning with Graph Networks Learn a Generalizable Branching Heuristic for a SAT Solver?","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-multi-agent-communication-through","slug":"learning-multi-agent-communication-through","title":"Learning Multi-Agent Communication through Structured Attentive Reasoning","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-maximum-entropy-inverse","slug":"revisiting-maximum-entropy-inverse","title":"Revisiting Maximum Entropy Inverse Reinforcement Learning: New Perspectives and Algorithms","date":"2020-12-01","arxiv_id":"2012.00889","repositories_listed":1,"syntology":null},{"url":"/paper/rl-unplugged-a-collection-of-benchmarks-for","slug":"rl-unplugged-a-collection-of-benchmarks-for","title":"RL Unplugged: A Collection of Benchmarks for Offline Reinforcement Learning","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/continuous-transition-improving-sample","slug":"continuous-transition-improving-sample","title":"Continuous Transition: Improving Sample Efficiency for Continuous Control Problems via MixUp","date":"2020-11-30","arxiv_id":"2011.14487","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-the-neural-architecture-of","slug":"optimizing-the-neural-architecture-of","title":"Optimizing the Neural Architecture of Reinforcement Learning Agents","date":"2020-11-30","arxiv_id":"2011.14632","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-visual-reinforcement-learning-1","slug":"self-supervised-visual-reinforcement-learning-1","title":"Self-supervised Visual Reinforcement Learning with Object-centric Representations","date":"2020-11-29","arxiv_id":"2011.14381","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-information-diffusion-in-time","slug":"efficient-information-diffusion-in-time","title":"Efficient Information Diffusion in Time-Varying Graphs through Deep Reinforcement Learning","date":"2020-11-27","arxiv_id":"2011.13518","repositories_listed":1,"syntology":null},{"url":"/paper/an-end-to-end-deep-reinforcement-learning","slug":"an-end-to-end-deep-reinforcement-learning","title":"An End-to-end Deep Reinforcement Learning Approach for the Long-term Short-term Planning on the Frenet Space","date":"2020-11-26","arxiv_id":"2011.13098","repositories_listed":1,"syntology":null},{"url":"/paper/interactive-machine-learning-of-musical","slug":"interactive-machine-learning-of-musical","title":"Interactive Machine Learning of Musical Gesture","date":"2020-11-26","arxiv_id":"2011.13487","repositories_listed":1,"syntology":null},{"url":"/paper/optimization-of-the-model-predictive-control","slug":"optimization-of-the-model-predictive-control","title":"Optimization of the Model Predictive Control Update Interval Using Reinforcement Learning","date":"2020-11-26","arxiv_id":"2011.13365","repositories_listed":1,"syntology":null},{"url":"/paper/accommodating-picky-customers-regret-bound","slug":"accommodating-picky-customers-regret-bound","title":"Accommodating Picky Customers: Regret Bound and Exploration Complexity for Multi-Objective Reinforcement Learning","date":"2020-11-25","arxiv_id":"2011.13034","repositories_listed":1,"syntology":null},{"url":"/paper/combining-semantic-guidance-and-deep","slug":"combining-semantic-guidance-and-deep","title":"Combining Semantic Guidance and Deep Reinforcement Learning For Generating Human Level Paintings","date":"2020-11-25","arxiv_id":"2011.12589","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-reinforcement-learning-is-a","slug":"distributed-reinforcement-learning-is-a","title":"RLlib Flow: Distributed Reinforcement Learning is a Dataflow Problem","date":"2020-11-25","arxiv_id":"2011.12719","repositories_listed":1,"syntology":null},{"url":"/paper/symmetry-aware-actor-critic-for-3d-molecular-1","slug":"symmetry-aware-actor-critic-for-3d-molecular-1","title":"Symmetry-Aware Actor-Critic for 3D Molecular Design","date":"2020-11-25","arxiv_id":"2011.12747","repositories_listed":1,"syntology":null},{"url":"/paper/tleague-a-framework-for-competitive-self-play","slug":"tleague-a-framework-for-competitive-self-play","title":"TLeague: A Framework for Competitive Self-Play based Distributed Multi-Agent Reinforcement Learning","date":"2020-11-25","arxiv_id":"2011.12895","repositories_listed":1,"syntology":null},{"url":"/paper/world-model-as-a-graph-learning-latent","slug":"world-model-as-a-graph-learning-latent","title":"World Model as a Graph: Learning Latent Landmarks for Planning","date":"2020-11-25","arxiv_id":"2011.12491","repositories_listed":1,"syntology":null},{"url":"/paper/learning-principle-of-least-action-with","slug":"learning-principle-of-least-action-with","title":"Learning Principle of Least Action with Reinforcement Learning","date":"2020-11-24","arxiv_id":"2011.11891","repositories_listed":1,"syntology":null},{"url":"/paper/an-analysis-of-reinforcement-learning-applied","slug":"an-analysis-of-reinforcement-learning-applied","title":"An analysis of Reinforcement Learning applied to Coach task in IEEE Very Small Size Soccer","date":"2020-11-23","arxiv_id":"2011.11785","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-representation-learning","slug":"an-empirical-study-of-representation-learning","title":"An Empirical Study of Representation Learning for Reinforcement Learning in Healthcare","date":"2020-11-23","arxiv_id":"2011.11235","repositories_listed":1,"syntology":null},{"url":"/paper/evolutionary-planning-in-latent-space","slug":"evolutionary-planning-in-latent-space","title":"Evolutionary Planning in Latent Space","date":"2020-11-23","arxiv_id":"2011.11293","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-feedback","slug":"deep-reinforcement-learning-for-feedback","title":"Deep reinforcement learning for feedback control in a collective flashing ratchet","date":"2020-11-20","arxiv_id":"2011.10357","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-exploration-for-model-based-1","slug":"efficient-exploration-for-model-based-1","title":"Model-based Reinforcement Learning for Continuous Control with Posterior Sampling","date":"2020-11-20","arxiv_id":"2012.09613","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/efficient-exploration-for-model-based-1#ran","syntology_url":"https://syntology.ai/paper/2012.09613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.09613"}},"official":{"repos":["yingfan-bot/mbpsrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inverse-constrained-reinforcement-learning","slug":"inverse-constrained-reinforcement-learning","title":"Inverse Constrained Reinforcement Learning","date":"2020-11-19","arxiv_id":"2011.09999","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/inverse-constrained-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2011.09999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.09999"}},"official":{"repos":["shehryar-malik/icrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"417b23a6631d578d6bccc71bd04741a4e095b05b10d090bd94d16a6939486484","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}