{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-2/papers/31","list_of":"/task/reinforcement-learning-2","task":"reinforcement-learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":31,"pages_in_order":135,"rows_per_page":100,"rows":[3001,3100],"of":13427,"counts":{"archive_papers_tagged":13427,"with_a_code_link":4119,"where_syntology_ran_a_sample":1165,"not_listed_spam_title":0,"listed":13427,"listed_where_code_ran":1165,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":973,"every_run_a_failure_of_syntologys_instrument":192,"listed_with_a_run_with_no_instrument_failure":973,"listed_every_run_a_failure_of_syntologys_instrument":192,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-2","prev":"/task/reinforcement-learning-2/papers/30","next":"/task/reinforcement-learning-2/papers/32","papers":[{"url":"/paper/does-standard-backpropagation-forget-less","slug":"does-standard-backpropagation-forget-less","title":"Does the Adam Optimizer Exacerbate Catastrophic Forgetting?","date":"2021-02-15","arxiv_id":"2102.07686","repositories_listed":1,"syntology":null},{"url":"/paper/how-rl-agents-behave-when-their-actions-are","slug":"how-rl-agents-behave-when-their-actions-are","title":"How RL Agents Behave When Their Actions Are Modified","date":"2021-02-15","arxiv_id":"2102.07716","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-electric-vehicle-charging","slug":"intelligent-electric-vehicle-charging","title":"Intelligent Electric Vehicle Charging Recommendation Based on Multi-Agent Reinforcement Learning","date":"2021-02-15","arxiv_id":"2102.07359","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-multi-agent-reinforcement-learning","slug":"scaling-multi-agent-reinforcement-learning","title":"Scaling Multi-Agent Reinforcement Learning with Selective Parameter Sharing","date":"2021-02-15","arxiv_id":"2102.07475","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/scaling-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2102.07475","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.07475"}},"official":{"repos":["uoe-agents/seps"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scrofazero-mastering-trick-taking-poker-game","slug":"scrofazero-mastering-trick-taking-poker-game","title":"ScrofaZero: Mastering Trick-taking Poker Game Gongzhu by Deep Reinforcement Learning","date":"2021-02-15","arxiv_id":"2102.07495","repositories_listed":1,"syntology":null},{"url":"/paper/q-value-weighted-regression-reinforcement-1","slug":"q-value-weighted-regression-reinforcement-1","title":"Q-Value Weighted Regression: Reinforcement Learning with Limited Data","date":"2021-02-12","arxiv_id":"2102.06782","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/q-value-weighted-regression-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2102.06782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.06782"}},"official":null}},{"url":"/paper/domain-adaptation-in-reinforcement-learning","slug":"domain-adaptation-in-reinforcement-learning","title":"Domain Adaptation In Reinforcement Learning Via Latent Unified State Representation","date":"2021-02-10","arxiv_id":"2102.05714","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":3,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/domain-adaptation-in-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2102.05714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.05714"}},"official":{"repos":["KarlXing/LUSR"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-augmentation-an-exploration-strategy","slug":"policy-augmentation-an-exploration-strategy","title":"Policy Augmentation: An Exploration Strategy for Faster Convergence of Deep Reinforcement Learning Algorithms","date":"2021-02-10","arxiv_id":"2102.05249","repositories_listed":1,"syntology":null},{"url":"/paper/risk-averse-offline-reinforcement-learning-1","slug":"risk-averse-offline-reinforcement-learning-1","title":"Risk-Averse Offline Reinforcement Learning","date":"2021-02-10","arxiv_id":"2102.05371","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-time-model-based-reinforcement","slug":"continuous-time-model-based-reinforcement","title":"Continuous-Time Model-Based Reinforcement Learning","date":"2021-02-09","arxiv_id":"2102.04764","repositories_listed":1,"syntology":null},{"url":"/paper/rl-reach-reproducible-reinforcement-learning","slug":"rl-reach-reproducible-reinforcement-learning","title":"rl_reach: Reproducible Reinforcement Learning Experiments for Robotic Reaching Tasks","date":"2021-02-09","arxiv_id":"2102.04916","repositories_listed":1,"syntology":null},{"url":"/paper/grid-to-graph-flexible-spatial-relational","slug":"grid-to-graph-flexible-spatial-relational","title":"Grid-to-Graph: Flexible Spatial Relational Inductive Biases for Reinforcement Learning","date":"2021-02-08","arxiv_id":"2102.04220","repositories_listed":1,"syntology":null},{"url":"/paper/rl-scope-cross-stack-profiling-for-deep","slug":"rl-scope-cross-stack-profiling-for-deep","title":"RL-Scope: Cross-Stack Profiling for Deep Reinforcement Learning Workloads","date":"2021-02-08","arxiv_id":"2102.04285","repositories_listed":1,"syntology":null},{"url":"/paper/sparsely-ensembled-convolutional-neural","slug":"sparsely-ensembled-convolutional-neural","title":"Sparsely ensembled convolutional neural network classifiers via reinforcement learning","date":"2021-02-07","arxiv_id":"2102.03921","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-reinforcement-learning-for","slug":"explainable-reinforcement-learning-for","title":"Explainable Reinforcement Learning for Longitudinal Control","date":"2021-02-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/longicontrol-a-reinforcement-learning","slug":"longicontrol-a-reinforcement-learning","title":"LongiControl: A Reinforcement Learning Environment for Longitudinal Vehicle Control","date":"2021-02-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-smart","slug":"deep-reinforcement-learning-for-smart","title":"Deep reinforcement learning for smart calibration of radio telescopes","date":"2021-02-05","arxiv_id":"2102.03200","repositories_listed":1,"syntology":null},{"url":"/paper/gnn-rl-compression-topology-aware-network","slug":"gnn-rl-compression-topology-aware-network","title":"Topology-Aware Network Pruning using Multi-stage Graph Embedding and Reinforcement Learning","date":"2021-02-05","arxiv_id":"2102.03214","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gnn-rl-compression-topology-aware-network#ran","syntology_url":"https://syntology.ai/paper/2102.03214","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.03214"}},"official":{"repos":["yusx-swapp/gnn-rl-model-compression"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/alchemy-a-structured-task-distribution-for","slug":"alchemy-a-structured-task-distribution-for","title":"Alchemy: A benchmark and analysis toolkit for meta-reinforcement learning agents","date":"2021-02-04","arxiv_id":"2102.02926","repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-with","slug":"multi-agent-reinforcement-learning-with","title":"Multi-Agent Reinforcement Learning with Temporal Logic Specifications","date":"2021-02-01","arxiv_id":"2102.00582","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/multi-agent-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2102.00582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.00582"}},"official":{"repos":["lrhammond/almanac"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/variation-resistant-q-learning-controlling","slug":"variation-resistant-q-learning-controlling","title":"Variation-resistant Q-learning: Controlling and Utilizing Estimation Bias in Reinforcement Learning for Better Performance","date":"2021-02-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/meta-reinforcement-learning-for-reliable","slug":"meta-reinforcement-learning-for-reliable","title":"Meta-Reinforcement Learning for Reliable Communication in THz/VLC Wireless VR Networks","date":"2021-01-29","arxiv_id":"2102.12277","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-the-impact-of-tunable-agents-in","slug":"exploring-the-impact-of-tunable-agents-in","title":"Exploring the Impact of Tunable Agents in Sequential Social Dilemmas","date":"2021-01-28","arxiv_id":"2101.11967","repositories_listed":1,"syntology":null},{"url":"/paper/data-sharing-games","slug":"data-sharing-games","title":"Data sharing games","date":"2021-01-26","arxiv_id":"2101.10721","repositories_listed":1,"syntology":null},{"url":"/paper/learning-synthetic-environments-for","slug":"learning-synthetic-environments-for","title":"Learning Synthetic Environments for Reinforcement Learning with Evolution Strategies","date":"2021-01-24","arxiv_id":"2101.09721","repositories_listed":1,"syntology":null},{"url":"/paper/differentiable-trust-region-layers-for-deep-1","slug":"differentiable-trust-region-layers-for-deep-1","title":"Differentiable Trust Region Layers for Deep Reinforcement Learning","date":"2021-01-22","arxiv_id":"2101.09207","repositories_listed":1,"syntology":null},{"url":"/paper/theory-of-mind-for-deep-reinforcement","slug":"theory-of-mind-for-deep-reinforcement","title":"Theory of Mind for Deep Reinforcement Learning in Hanabi","date":"2021-01-22","arxiv_id":"2101.09328","repositories_listed":1,"syntology":null},{"url":"/paper/mt5b3-a-framework-for-building","slug":"mt5b3-a-framework-for-building","title":"mt5se: An Open Source Framework for Building Autonomous Trading Robots","date":"2021-01-20","arxiv_id":"2101.08169","repositories_listed":1,"syntology":null},{"url":"/paper/updet-universal-multi-agent-reinforcement","slug":"updet-universal-multi-agent-reinforcement","title":"UPDeT: Universal Multi-agent Reinforcement Learning via Policy Decoupling with Transformers","date":"2021-01-20","arxiv_id":"2101.08001","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-producing","slug":"deep-reinforcement-learning-for-producing","title":"Deep Reinforcement Learning for Producing Furniture Layout in Indoor Scenes","date":"2021-01-19","arxiv_id":"2101.07462","repositories_listed":1,"syntology":null},{"url":"/paper/grounding-language-to-entities-and-dynamics","slug":"grounding-language-to-entities-and-dynamics","title":"Grounding Language to Entities and Dynamics for Generalization in Reinforcement Learning","date":"2021-01-19","arxiv_id":"2101.07393","repositories_listed":1,"syntology":null},{"url":"/paper/towards-facilitating-empathic-conversations","slug":"towards-facilitating-empathic-conversations","title":"Towards Facilitating Empathic Conversations in Online Mental Health Support: A Reinforcement Learning Approach","date":"2021-01-19","arxiv_id":"2101.07714","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-active-high","slug":"deep-reinforcement-learning-for-active-high","title":"Deep Reinforcement Learning for Active High Frequency Trading","date":"2021-01-18","arxiv_id":"2101.07107","repositories_listed":1,"syntology":null},{"url":"/paper/hammer-multi-level-coordination-of","slug":"hammer-multi-level-coordination-of","title":"HAMMER: Multi-Level Coordination of Reinforcement Learning Agents via Learned Messaging","date":"2021-01-18","arxiv_id":"2102.00824","repositories_listed":1,"syntology":null},{"url":"/paper/interpretable-policy-specification-and","slug":"interpretable-policy-specification-and","title":"Natural Language Specification of Reinforcement Learning Policies through Differentiable Decision Trees","date":"2021-01-18","arxiv_id":"2101.07140","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-by-1","slug":"hierarchical-reinforcement-learning-by-1","title":"Hierarchical Reinforcement Learning By Discovering Intrinsic Options","date":"2021-01-16","arxiv_id":"2101.06521","repositories_listed":1,"syntology":null},{"url":"/paper/controlling-the-risk-of-conversational-search","slug":"controlling-the-risk-of-conversational-search","title":"Controlling the Risk of Conversational Search via Reinforcement Learning","date":"2021-01-15","arxiv_id":"2101.06327","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-behavioral-similarity-embeddings-1","slug":"contrastive-behavioral-similarity-embeddings-1","title":"Contrastive Behavioral Similarity Embeddings for Generalization in Reinforcement Learning","date":"2021-01-13","arxiv_id":"2101.05265","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-soccer-player-from-live-camera-to","slug":"evaluating-soccer-player-from-live-camera-to","title":"Evaluating Soccer Player: from Live Camera to Deep Reinforcement Learning","date":"2021-01-13","arxiv_id":"2101.05388","repositories_listed":1,"syntology":null},{"url":"/paper/memory-augmented-reinforcement-learning-for","slug":"memory-augmented-reinforcement-learning-for","title":"Memory-Augmented Reinforcement Learning for Image-Goal Navigation","date":"2021-01-13","arxiv_id":"2101.05181","repositories_listed":1,"syntology":null},{"url":"/paper/action-priors-for-large-action-spaces-in","slug":"action-priors-for-large-action-spaces-in","title":"Action Priors for Large Action Spaces in Robotics","date":"2021-01-11","arxiv_id":"2101.04178","repositories_listed":1,"syntology":null},{"url":"/paper/implicit-unlikelihood-training-improving","slug":"implicit-unlikelihood-training-improving","title":"Implicit Unlikelihood Training: Improving Neural Text Generation with Reinforcement Learning","date":"2021-01-11","arxiv_id":"2101.04229","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-function","slug":"deep-reinforcement-learning-with-function","title":"Deep Reinforcement Learning with Function Properties in Mean Reversion Strategies","date":"2021-01-09","arxiv_id":"2101.03418","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-based-encoder","slug":"a-reinforcement-learning-based-encoder","title":"A Reinforcement Learning Based Encoder-Decoder Framework for Learning Stock Trading Rules","date":"2021-01-08","arxiv_id":"2101.03867","repositories_listed":1,"syntology":null},{"url":"/paper/simulating-sql-injection-vulnerability","slug":"simulating-sql-injection-vulnerability","title":"Simulating SQL Injection Vulnerability Exploitation Using Q-Learning Reinforcement Learning Agents","date":"2021-01-08","arxiv_id":"2101.03118","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-collective","slug":"reinforcement-learning-based-collective","title":"Reinforcement Learning based Collective Entity Alignment with Adaptive Features","date":"2021-01-05","arxiv_id":"2101.01353","repositories_listed":1,"syntology":null},{"url":"/paper/a-novel-policy-for-pre-trained-deep","slug":"a-novel-policy-for-pre-trained-deep","title":"A novel policy for pre-trained Deep Reinforcement Learning for Speech Emotion Recognition","date":"2021-01-04","arxiv_id":"2101.00738","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-domain-adaptation-for","slug":"cross-modal-domain-adaptation-for","title":"Cross-Modal Domain Adaptation for Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/faults-in-deep-reinforcement-learning","slug":"faults-in-deep-reinforcement-learning","title":"Faults in Deep Reinforcement Learning Programs: A Taxonomy and A Detection Approach","date":"2021-01-01","arxiv_id":"2101.00135","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-meta-reinforcement-learning-for","slug":"hierarchical-meta-reinforcement-learning-for","title":"Hierarchical Meta Reinforcement Learning for Multi-Task Environments","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-trust-region-learning","slug":"multi-agent-trust-region-learning","title":"Multi-Agent Trust Region Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/structure-and-randomness-in-planning-and","slug":"structure-and-randomness-in-planning-and","title":"Structure and randomness in planning and reinforcement learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/trust-but-verify-model-based-exploration-in","slug":"trust-but-verify-model-based-exploration-in","title":"Trust, but verify: model-based exploration in sparse reward environments","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/unsupervised-task-clustering-for-multi-task","slug":"unsupervised-task-clustering-for-multi-task","title":"Unsupervised Task Clustering for Multi-Task Reinforcement Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/multi-agent-reinforcement-learning-for-7","slug":"multi-agent-reinforcement-learning-for-7","title":"Multi-Agent Reinforcement Learning for Unmanned Aerial Vehicle Coordination by Multi-Critic Policy Gradient Optimization","date":"2020-12-31","arxiv_id":"2012.15472","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-visual-planning-with-self-1","slug":"model-based-visual-planning-with-self-1","title":"Model-Based Visual Planning with Self-Supervised Functional Distances","date":"2020-12-30","arxiv_id":"2012.15373","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-portfolio","slug":"deep-reinforcement-learning-for-portfolio","title":"Deep Reinforcement Learning for Long-Short Portfolio Optimization","date":"2020-12-26","arxiv_id":"2012.13773","repositories_listed":1,"syntology":null},{"url":"/paper/qvmix-and-qvmix-max-extending-the-deep","slug":"qvmix-and-qvmix-max-extending-the-deep","title":"QVMix and QVMix-Max: Extending the Deep Quality-Value Family of Algorithms to Cooperative Multi-Agent Reinforcement Learning","date":"2020-12-22","arxiv_id":"2012.12062","repositories_listed":1,"syntology":null},{"url":"/paper/2012-11662","slug":"2012-11662","title":"Explicitly Encouraging Low Fractional Dimensional Trajectories Via Reinforcement Learning","date":"2020-12-21","arxiv_id":"2012.11662","repositories_listed":1,"syntology":null},{"url":"/paper/offline-reinforcement-learning-from-images","slug":"offline-reinforcement-learning-from-images","title":"Offline Reinforcement Learning from Images with Latent Space Models","date":"2020-12-21","arxiv_id":"2012.11547","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-joint-1","slug":"deep-reinforcement-learning-for-joint-1","title":"Deep Reinforcement Learning for Joint Spectrum and Power Allocation in Cellular Networks","date":"2020-12-19","arxiv_id":"2012.10682","repositories_listed":1,"syntology":null},{"url":"/paper/multi-decoder-attention-model-with-embedding","slug":"multi-decoder-attention-model-with-embedding","title":"Multi-Decoder Attention Model with Embedding Glimpse for Solving Vehicle Routing Problems","date":"2020-12-19","arxiv_id":"2012.10638","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-decoder-attention-model-with-embedding#ran","syntology_url":"https://syntology.ai/paper/2012.10638","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.10638"}},"official":{"repos":["liangxinedu/MDAM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/high-throughput-synchronous-deep-rl-1","slug":"high-throughput-synchronous-deep-rl-1","title":"High-Throughput Synchronous Deep RL","date":"2020-12-17","arxiv_id":"2012.09849","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/high-throughput-synchronous-deep-rl-1#ran","syntology_url":"https://syntology.ai/paper/2012.09849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.09849"}},"official":{"repos":["IouJenLiu/HTS-RL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-the-efficient-neural-architecture","slug":"improving-the-efficient-neural-architecture","title":"Improving the Efficient Neural Architecture Search via Rewarding Modifications","date":"2020-12-17","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/model-free-and-bayesian-ensembling-model","slug":"model-free-and-bayesian-ensembling-model","title":"Model-free and Bayesian Ensembling Model-based Deep Reinforcement Learning for Particle Accelerator Control Demonstrated on the FERMI FEL","date":"2020-12-17","arxiv_id":"2012.09737","repositories_listed":1,"syntology":null},{"url":"/paper/learning-accurate-long-term-dynamics-for","slug":"learning-accurate-long-term-dynamics-for","title":"Learning Accurate Long-term Dynamics for Model-based Reinforcement Learning","date":"2020-12-16","arxiv_id":"2012.09156","repositories_listed":1,"syntology":null},{"url":"/paper/cloud-database-tuning-with-reinforcement","slug":"cloud-database-tuning-with-reinforcement","title":"Cloud Database Tuning with Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/evolutionary-learning-of-interpretable","slug":"evolutionary-learning-of-interpretable","title":"Evolutionary learning of interpretable decision trees","date":"2020-12-14","arxiv_id":"2012.07723","repositories_listed":1,"syntology":null},{"url":"/paper/increasing-data-efficiency-of-driving-agent","slug":"increasing-data-efficiency-of-driving-agent","title":"Increasing Data Efficiency of Driving Agent By World Model","date":"2020-12-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/policy-gradient-rl-algorithms-as-directed","slug":"policy-gradient-rl-algorithms-as-directed","title":"Policy Gradient RL Algorithms as Directed Acyclic Graphs","date":"2020-12-14","arxiv_id":"2012.07763","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-contact-rich-tasks","slug":"reinforcement-learning-for-contact-rich-tasks","title":"Reinforcement Learning for Contact-Rich Tasks: Robotic Peg Insertion Strategies","date":"2020-12-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/sim-to-real-reinforcement-learning-applied-to","slug":"sim-to-real-reinforcement-learning-applied-to","title":"Sim-to-real reinforcement learning applied to end-to-end vehicle control","date":"2020-12-14","arxiv_id":"2012.07461","repositories_listed":1,"syntology":null},{"url":"/paper/super-reinforcement-bros-playing-super-mario","slug":"super-reinforcement-bros-playing-super-mario","title":"Super Reinforcement Bros: Playing Super Mario Bros with Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/an-efficient-asynchronous-method-for-1","slug":"an-efficient-asynchronous-method-for-1","title":"An Efficient Asynchronous Method for Integrating Evolutionary and Gradient-based Policy Search","date":"2020-12-10","arxiv_id":"2012.05417","repositories_listed":1,"syntology":null},{"url":"/paper/combining-reinforcement-learning-with-lin","slug":"combining-reinforcement-learning-with-lin","title":"Combining Reinforcement Learning with Lin-Kernighan-Helsgaun Algorithm for the Traveling Salesman Problem","date":"2020-12-08","arxiv_id":"2012.04461","repositories_listed":1,"syntology":null},{"url":"/paper/navrep-unsupervised-representations-for","slug":"navrep-unsupervised-representations-for","title":"NavRep: Unsupervised Representations for Reinforcement Learning of Robot Navigation in Dynamic Human Environments","date":"2020-12-08","arxiv_id":"2012.04406","repositories_listed":1,"syntology":null},{"url":"/paper/gaea-graph-augmentation-for-equitable-access","slug":"gaea-graph-augmentation-for-equitable-access","title":"GAEA: Graph Augmentation for Equitable Access via Reinforcement Learning","date":"2020-12-07","arxiv_id":"2012.03900","repositories_listed":1,"syntology":null},{"url":"/paper/rloc-terrain-aware-legged-locomotion-using","slug":"rloc-terrain-aware-legged-locomotion-using","title":"RLOC: Terrain-Aware Legged Locomotion using Reinforcement Learning and Optimal Control","date":"2020-12-05","arxiv_id":"2012.03094","repositories_listed":1,"syntology":null},{"url":"/paper/learning-multi-agent-communication-through","slug":"learning-multi-agent-communication-through","title":"Learning Multi-Agent Communication through Structured Attentive Reasoning","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/revisiting-maximum-entropy-inverse","slug":"revisiting-maximum-entropy-inverse","title":"Revisiting Maximum Entropy Inverse Reinforcement Learning: New Perspectives and Algorithms","date":"2020-12-01","arxiv_id":"2012.00889","repositories_listed":1,"syntology":null},{"url":"/paper/rl-unplugged-a-collection-of-benchmarks-for","slug":"rl-unplugged-a-collection-of-benchmarks-for","title":"RL Unplugged: A Collection of Benchmarks for Offline Reinforcement Learning","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-the-neural-architecture-of","slug":"optimizing-the-neural-architecture-of","title":"Optimizing the Neural Architecture of Reinforcement Learning Agents","date":"2020-11-30","arxiv_id":"2011.14632","repositories_listed":1,"syntology":null},{"url":"/paper/self-supervised-visual-reinforcement-learning-1","slug":"self-supervised-visual-reinforcement-learning-1","title":"Self-supervised Visual Reinforcement Learning with Object-centric Representations","date":"2020-11-29","arxiv_id":"2011.14381","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-information-diffusion-in-time","slug":"efficient-information-diffusion-in-time","title":"Efficient Information Diffusion in Time-Varying Graphs through Deep Reinforcement Learning","date":"2020-11-27","arxiv_id":"2011.13518","repositories_listed":1,"syntology":null},{"url":"/paper/an-end-to-end-deep-reinforcement-learning","slug":"an-end-to-end-deep-reinforcement-learning","title":"An End-to-end Deep Reinforcement Learning Approach for the Long-term Short-term Planning on the Frenet Space","date":"2020-11-26","arxiv_id":"2011.13098","repositories_listed":1,"syntology":null},{"url":"/paper/optimization-of-the-model-predictive-control","slug":"optimization-of-the-model-predictive-control","title":"Optimization of the Model Predictive Control Update Interval Using Reinforcement Learning","date":"2020-11-26","arxiv_id":"2011.13365","repositories_listed":1,"syntology":null},{"url":"/paper/accommodating-picky-customers-regret-bound","slug":"accommodating-picky-customers-regret-bound","title":"Accommodating Picky Customers: Regret Bound and Exploration Complexity for Multi-Objective Reinforcement Learning","date":"2020-11-25","arxiv_id":"2011.13034","repositories_listed":1,"syntology":null},{"url":"/paper/combining-semantic-guidance-and-deep","slug":"combining-semantic-guidance-and-deep","title":"Combining Semantic Guidance and Deep Reinforcement Learning For Generating Human Level Paintings","date":"2020-11-25","arxiv_id":"2011.12589","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-reinforcement-learning-is-a","slug":"distributed-reinforcement-learning-is-a","title":"RLlib Flow: Distributed Reinforcement Learning is a Dataflow Problem","date":"2020-11-25","arxiv_id":"2011.12719","repositories_listed":1,"syntology":null},{"url":"/paper/symmetry-aware-actor-critic-for-3d-molecular-1","slug":"symmetry-aware-actor-critic-for-3d-molecular-1","title":"Symmetry-Aware Actor-Critic for 3D Molecular Design","date":"2020-11-25","arxiv_id":"2011.12747","repositories_listed":1,"syntology":null},{"url":"/paper/tleague-a-framework-for-competitive-self-play","slug":"tleague-a-framework-for-competitive-self-play","title":"TLeague: A Framework for Competitive Self-Play based Distributed Multi-Agent Reinforcement Learning","date":"2020-11-25","arxiv_id":"2011.12895","repositories_listed":1,"syntology":null},{"url":"/paper/world-model-as-a-graph-learning-latent","slug":"world-model-as-a-graph-learning-latent","title":"World Model as a Graph: Learning Latent Landmarks for Planning","date":"2020-11-25","arxiv_id":"2011.12491","repositories_listed":1,"syntology":null},{"url":"/paper/learning-principle-of-least-action-with","slug":"learning-principle-of-least-action-with","title":"Learning Principle of Least Action with Reinforcement Learning","date":"2020-11-24","arxiv_id":"2011.11891","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-representation-learning","slug":"an-empirical-study-of-representation-learning","title":"An Empirical Study of Representation Learning for Reinforcement Learning in Healthcare","date":"2020-11-23","arxiv_id":"2011.11235","repositories_listed":1,"syntology":null},{"url":"/paper/evolutionary-planning-in-latent-space","slug":"evolutionary-planning-in-latent-space","title":"Evolutionary Planning in Latent Space","date":"2020-11-23","arxiv_id":"2011.11293","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-feedback","slug":"deep-reinforcement-learning-for-feedback","title":"Deep reinforcement learning for feedback control in a collective flashing ratchet","date":"2020-11-20","arxiv_id":"2011.10357","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-exploration-for-model-based-1","slug":"efficient-exploration-for-model-based-1","title":"Model-based Reinforcement Learning for Continuous Control with Posterior Sampling","date":"2020-11-20","arxiv_id":"2012.09613","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/efficient-exploration-for-model-based-1#ran","syntology_url":"https://syntology.ai/paper/2012.09613","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.09613"}},"official":{"repos":["yingfan-bot/mbpsrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inverse-constrained-reinforcement-learning","slug":"inverse-constrained-reinforcement-learning","title":"Inverse Constrained Reinforcement Learning","date":"2020-11-19","arxiv_id":"2011.09999","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/inverse-constrained-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2011.09999","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.09999"}},"official":{"repos":["shehryar-malik/icrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/combining-reinforcement-learning-with-model","slug":"combining-reinforcement-learning-with-model","title":"Combining Reinforcement Learning with Model Predictive Control for On-Ramp Merging","date":"2020-11-17","arxiv_id":"2011.08484","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-clustering-in-particle-physics","slug":"hierarchical-clustering-in-particle-physics","title":"Hierarchical clustering in particle physics through reinforcement learning","date":"2020-11-16","arxiv_id":"2011.08191","repositories_listed":1,"syntology":null}],"record_sha256":"9cf637e030a00feebf83d58fe4cff0f000fd0033c067801c0aa489d311bf4bb0","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}