{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/reinforcement-learning-1/papers/29","list_of":"/task/reinforcement-learning-1","task":"Reinforcement Learning (RL)","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":29,"pages_in_order":152,"rows_per_page":100,"rows":[2801,2900],"of":15113,"counts":{"archive_papers_tagged":15113,"with_a_code_link":4749,"where_syntology_ran_a_sample":1416,"not_listed_spam_title":0,"listed":15113,"listed_where_code_ran":1416,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1186,"every_run_a_failure_of_syntologys_instrument":230,"listed_with_a_run_with_no_instrument_failure":1186,"listed_every_run_a_failure_of_syntologys_instrument":230,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/reinforcement-learning-1","prev":"/task/reinforcement-learning-1/papers/28","next":"/task/reinforcement-learning-1/papers/30","papers":[{"url":"/paper/pearl-parallel-evolutionary-and-reinforcement","slug":"pearl-parallel-evolutionary-and-reinforcement","title":"Pearl: Parallel Evolutionary and Reinforcement Learning Library","date":"2022-01-24","arxiv_id":"2201.09568","repositories_listed":1,"syntology":null},{"url":"/paper/the-paradox-of-choice-using-attention-in","slug":"the-paradox-of-choice-using-attention-in","title":"The Paradox of Choice: Using Attention in Hierarchical Reinforcement Learning","date":"2022-01-24","arxiv_id":"2201.09653","repositories_listed":1,"syntology":null},{"url":"/paper/bag-of-tricks-for-natural-policy-gradient","slug":"bag-of-tricks-for-natural-policy-gradient","title":"Understanding the Effects of Second-Order Approximations in Natural Policy Gradient Reinforcement Learning","date":"2022-01-22","arxiv_id":"2201.09104","repositories_listed":1,"syntology":null},{"url":"/paper/environment-generation-for-zero-shot-1","slug":"environment-generation-for-zero-shot-1","title":"Environment Generation for Zero-Shot Compositional Reinforcement Learning","date":"2022-01-21","arxiv_id":"2201.08896","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/environment-generation-for-zero-shot-1#ran","syntology_url":"https://syntology.ai/paper/2201.08896","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.08896"}},"official":{"repos":["google-research/google-research"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tensor-and-matrix-low-rank-value-function","slug":"tensor-and-matrix-low-rank-value-function","title":"Tensor and Matrix Low-Rank Value-Function Approximation in Reinforcement Learning","date":"2022-01-21","arxiv_id":"2201.09736","repositories_listed":1,"syntology":null},{"url":"/paper/dropo-sim-to-real-transfer-with-offline","slug":"dropo-sim-to-real-transfer-with-offline","title":"DROPO: Sim-to-Real Transfer with Offline Domain Randomization","date":"2022-01-20","arxiv_id":"2201.08434","repositories_listed":1,"syntology":null},{"url":"/paper/goal-conditioned-reinforcement-learning","slug":"goal-conditioned-reinforcement-learning","title":"Goal-Conditioned Reinforcement Learning: Problems and Solutions","date":"2022-01-20","arxiv_id":"2201.08299","repositories_listed":1,"syntology":null},{"url":"/paper/two-sample-testing-in-reinforcement-learning","slug":"two-sample-testing-in-reinforcement-learning","title":"Addressing Maximization Bias in Reinforcement Learning with Two-Sample Testing","date":"2022-01-20","arxiv_id":"2201.08078","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-textbook","slug":"reinforcement-learning-textbook","title":"Reinforcement Learning Textbook","date":"2022-01-19","arxiv_id":"2201.09746","repositories_listed":1,"syntology":null},{"url":"/paper/squire-a-sequence-to-sequence-framework-for","slug":"squire-a-sequence-to-sequence-framework-for","title":"SQUIRE: A Sequence-to-sequence Framework for Multi-hop Knowledge Graph Reasoning","date":"2022-01-17","arxiv_id":"2201.06206","repositories_listed":1,"syntology":null},{"url":"/paper/comparing-model-free-and-model-based","slug":"comparing-model-free-and-model-based","title":"Comparing Model-free and Model-based Algorithms for Offline Reinforcement Learning","date":"2022-01-14","arxiv_id":"2201.05433","repositories_listed":1,"syntology":null},{"url":"/paper/smart-magnetic-microrobots-learn-to-swim-with","slug":"smart-magnetic-microrobots-learn-to-swim-with","title":"Smart Magnetic Microrobots Learn to Swim with Deep Reinforcement Learning","date":"2022-01-14","arxiv_id":"2201.05599","repositories_listed":1,"syntology":null},{"url":"/paper/solving-dynamic-graph-problems-with-multi","slug":"solving-dynamic-graph-problems-with-multi","title":"Solving Dynamic Graph Problems with Multi-Attention Deep Reinforcement Learning","date":"2022-01-13","arxiv_id":"2201.04895","repositories_listed":1,"syntology":null},{"url":"/paper/weakly-supervised-scene-text-detection-using","slug":"weakly-supervised-scene-text-detection-using","title":"Weakly Supervised Scene Text Detection using Deep Reinforcement Learning","date":"2022-01-13","arxiv_id":"2201.04866","repositories_listed":1,"syntology":null},{"url":"/paper/agent-temporal-attention-for-reward","slug":"agent-temporal-attention-for-reward","title":"Agent-Temporal Attention for Reward Redistribution in Episodic Multi-Agent Reinforcement Learning","date":"2022-01-12","arxiv_id":"2201.04612","repositories_listed":1,"syntology":null},{"url":"/paper/in-defense-of-the-unitary-scalarization-for","slug":"in-defense-of-the-unitary-scalarization-for","title":"In Defense of the Unitary Scalarization for Deep Multi-Task Learning","date":"2022-01-11","arxiv_id":"2201.04122","repositories_listed":1,"syntology":null},{"url":"/paper/verified-probabilistic-policies-for-deep","slug":"verified-probabilistic-policies-for-deep","title":"Verified Probabilistic Policies for Deep Reinforcement Learning","date":"2022-01-10","arxiv_id":"2201.03698","repositories_listed":1,"syntology":null},{"url":"/paper/mirror-learning-a-unifying-framework-of","slug":"mirror-learning-a-unifying-framework-of","title":"Mirror Learning: A Unifying Framework of Policy Optimisation","date":"2022-01-07","arxiv_id":"2201.02373","repositories_listed":1,"syntology":null},{"url":"/paper/sablas-learning-safe-control-for-black-box","slug":"sablas-learning-safe-control-for-black-box","title":"SABLAS: Learning Safe Control for Black-box Dynamical Systems","date":"2022-01-06","arxiv_id":"2201.01918","repositories_listed":1,"syntology":null},{"url":"/paper/sample-efficient-deep-reinforcement-learning-5","slug":"sample-efficient-deep-reinforcement-learning-5","title":"Sample Efficient Deep Reinforcement Learning via Uncertainty Estimation","date":"2022-01-05","arxiv_id":"2201.01666","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-efficient-deep-reinforcement-learning-5#ran","syntology_url":"https://syntology.ai/paper/2201.01666","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.01666"}},"official":{"repos":["montrealrobotics/iv_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-learning-based-predictive-control-of","slug":"deep-learning-based-predictive-control-of","title":"Deep Learning-based Predictive Control of Battery Management for Frequency Regulation","date":"2022-01-04","arxiv_id":"2201.01166","repositories_listed":1,"syntology":null},{"url":"/paper/using-simulation-optimization-to-improve-zero","slug":"using-simulation-optimization-to-improve-zero","title":"Using Simulation Optimization to Improve Zero-shot Policy Transfer of Quadrotors","date":"2022-01-04","arxiv_id":"2201.01369","repositories_listed":1,"syntology":null},{"url":"/paper/hybrid-intelligence-for-dynamic-job-shop","slug":"hybrid-intelligence-for-dynamic-job-shop","title":"Hybrid intelligence for dynamic job-shop scheduling with deep reinforcement learning and attention mechanism","date":"2022-01-03","arxiv_id":"2201.00548","repositories_listed":1,"syntology":null},{"url":"/paper/toward-causal-aware-rl-state-wise-action","slug":"toward-causal-aware-rl-state-wise-action","title":"Toward Causal-Aware RL: State-Wise Action-Refined Temporal Difference","date":"2022-01-02","arxiv_id":"2201.00354","repositories_listed":1,"syntology":null},{"url":"/paper/settling-the-bias-and-variance-of-meta","slug":"settling-the-bias-and-variance-of-meta","title":"A Theoretical Understanding of Gradient Bias in Meta-Reinforcement Learning","date":"2021-12-31","arxiv_id":"2112.15400","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/settling-the-bias-and-variance-of-meta#ran","syntology_url":"https://syntology.ai/paper/2112.15400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.15400"}},"official":{"repos":["Benjamin-eecs/Theoretical-GMRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/constraint-sampling-reinforcement-learning","slug":"constraint-sampling-reinforcement-learning","title":"Constraint Sampling Reinforcement Learning: Incorporating Expertise For Faster Learning","date":"2021-12-30","arxiv_id":"2112.15221","repositories_listed":1,"syntology":null},{"url":"/paper/moral-aligning-ai-with-human-norms-through","slug":"moral-aligning-ai-with-human-norms-through","title":"MORAL: Aligning AI with Human Norms through Multi-Objective Reinforced Active Learning","date":"2021-12-30","arxiv_id":"2201.00012","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/moral-aligning-ai-with-human-norms-through#ran","syntology_url":"https://syntology.ai/paper/2201.00012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.00012"}},"official":{"repos":["mlpeschl/moral_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sequential-episodic-control","slug":"sequential-episodic-control","title":"Sequential memory improves sample and memory efficiency in Episodic Control","date":"2021-12-29","arxiv_id":"2112.14734","repositories_listed":1,"syntology":null},{"url":"/paper/exponential-family-model-based-reinforcement","slug":"exponential-family-model-based-reinforcement","title":"Exponential Family Model-Based Reinforcement Learning via Score Matching","date":"2021-12-28","arxiv_id":"2112.14195","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/exponential-family-model-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2112.14195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.14195"}},"official":{"repos":["anmolkabra/score-matching-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-the-performance-of-backward-chained","slug":"improving-the-performance-of-backward-chained","title":"Improving the Performance of Backward Chained Behavior Trees that use Reinforcement Learning","date":"2021-12-27","arxiv_id":"2112.13744","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-traffic-light-via-policy-based","slug":"intelligent-traffic-light-via-policy-based","title":"Intelligent Traffic Light via Policy-based Deep Reinforcement Learning","date":"2021-12-27","arxiv_id":"2112.13817","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-dynamic-convex","slug":"reinforcement-learning-with-dynamic-convex","title":"Reinforcement Learning with Dynamic Convex Risk Measures","date":"2021-12-26","arxiv_id":"2112.13414","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/reinforcement-learning-with-dynamic-convex#ran","syntology_url":"https://syntology.ai/paper/2112.13414","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.13414"}},"official":{"repos":["acoache/rl-dynamicconvexrisk"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-efficient-combinatorial-optimization-model","slug":"an-efficient-combinatorial-optimization-model","title":"An Efficient Combinatorial Optimization Model Using Learning-to-Rank Distillation","date":"2021-12-24","arxiv_id":"2201.00695","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-walk-with-dual-agents-for","slug":"learning-to-walk-with-dual-agents-for","title":"Learning to Walk with Dual Agents for Knowledge Graph Reasoning","date":"2021-12-23","arxiv_id":"2112.12876","repositories_listed":1,"syntology":null},{"url":"/paper/safety-and-liveness-guarantees-through-reach","slug":"safety-and-liveness-guarantees-through-reach","title":"Safety and Liveness Guarantees through Reach-Avoid Reinforcement Learning","date":"2021-12-23","arxiv_id":"2112.12288","repositories_listed":1,"syntology":null},{"url":"/paper/a-deep-reinforcement-learning-approach-for-11","slug":"a-deep-reinforcement-learning-approach-for-11","title":"A Deep Reinforcement Learning Approach for Solving the Traveling Salesman Problem with Drone","date":"2021-12-22","arxiv_id":"2112.12545","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-deep-reinforcement-learning-for","slug":"adversarial-deep-reinforcement-learning-for","title":"Adversarial Deep Reinforcement Learning for Improving the Robustness of Multi-agent Autonomous Driving Policies","date":"2021-12-22","arxiv_id":"2112.11937","repositories_listed":1,"syntology":null},{"url":"/paper/alpha-mini-minichess-agent-with-deep","slug":"alpha-mini-minichess-agent-with-deep","title":"Alpha-Mini: Minichess Agent with Deep Reinforcement Learning","date":"2021-12-22","arxiv_id":"2112.13666","repositories_listed":1,"syntology":null},{"url":"/paper/direct-behavior-specification-via-constrained","slug":"direct-behavior-specification-via-constrained","title":"Direct Behavior Specification via Constrained Reinforcement Learning","date":"2021-12-22","arxiv_id":"2112.12228","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-the-robustness-of-deep","slug":"evaluating-the-robustness-of-deep","title":"Evaluating the Robustness of Deep Reinforcement Learning for Autonomous Policies in a Multi-agent Urban Driving Environment","date":"2021-12-22","arxiv_id":"2112.11947","repositories_listed":1,"syntology":null},{"url":"/paper/newsvendor-model-with-deep-reinforcement","slug":"newsvendor-model-with-deep-reinforcement","title":"Newsvendor Model with Deep Reinforcement Learning","date":"2021-12-22","arxiv_id":"2112.12544","repositories_listed":1,"syntology":null},{"url":"/paper/off-environment-evaluation-using-convex-risk","slug":"off-environment-evaluation-using-convex-risk","title":"Off Environment Evaluation Using Convex Risk Minimization","date":"2021-12-21","arxiv_id":"2112.11532","repositories_listed":1,"syntology":null},{"url":"/paper/soft-actor-critic-with-cross-entropy-policy","slug":"soft-actor-critic-with-cross-entropy-policy","title":"Soft Actor-Critic with Cross-Entropy Policy Optimization","date":"2021-12-21","arxiv_id":"2112.11115","repositories_listed":1,"syntology":null},{"url":"/paper/differentially-private-regret-minimization-in","slug":"differentially-private-regret-minimization-in","title":"Differentially Private Regret Minimization in Episodic Markov Decision Processes","date":"2021-12-20","arxiv_id":"2112.10599","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/differentially-private-regret-minimization-in#ran","syntology_url":"https://syntology.ai/paper/2112.10599","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.10599"}},"official":{"repos":["xingyuzhou989/privatetabularrl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/variational-quantum-soft-actor-critic","slug":"variational-quantum-soft-actor-critic","title":"Variational Quantum Soft Actor-Critic","date":"2021-12-20","arxiv_id":"2112.11921","repositories_listed":1,"syntology":null},{"url":"/paper/expression-is-enough-improving-traffic-signal","slug":"expression-is-enough-improving-traffic-signal","title":"Expression might be enough: representing pressure and demand for reinforcement learning based traffic signal control","date":"2021-12-19","arxiv_id":"2112.10107","repositories_listed":1,"syntology":null},{"url":"/paper/exploiting-expert-guided-symmetry-detection","slug":"exploiting-expert-guided-symmetry-detection","title":"Data Augmentation through Expert-guided Symmetry Detection to Improve Performance in Offline Reinforcement Learning","date":"2021-12-18","arxiv_id":"2112.09943","repositories_listed":1,"syntology":null},{"url":"/paper/space-non-cooperative-object-active-tracking","slug":"space-non-cooperative-object-active-tracking","title":"Space Non-cooperative Object Active Tracking with Deep Reinforcement Learning","date":"2021-12-18","arxiv_id":"2112.09854","repositories_listed":1,"syntology":null},{"url":"/paper/distillation-of-rl-policies-with-formal","slug":"distillation-of-rl-policies-with-formal","title":"Distillation of RL Policies with Formal Guarantees via Variational Abstraction of Markov Decision Processes (Technical Report)","date":"2021-12-17","arxiv_id":"2112.09655","repositories_listed":1,"syntology":null},{"url":"/paper/inherently-explainable-reinforcement-learning","slug":"inherently-explainable-reinforcement-learning","title":"Inherently Explainable Reinforcement Learning in Natural Language","date":"2021-12-16","arxiv_id":"2112.08907","repositories_listed":1,"syntology":null},{"url":"/paper/feature-attending-recurrent-modules-for","slug":"feature-attending-recurrent-modules-for","title":"Feature-Attending Recurrent Modules for Generalization in Reinforcement Learning","date":"2021-12-15","arxiv_id":"2112.08369","repositories_listed":1,"syntology":null},{"url":"/paper/cem-gd-cross-entropy-method-with-gradient","slug":"cem-gd-cross-entropy-method-with-gradient","title":"CEM-GD: Cross-Entropy Method with Gradient Descent Planner for Model-Based Reinforcement Learning","date":"2021-12-14","arxiv_id":"2112.07746","repositories_listed":1,"syntology":null},{"url":"/paper/conjugated-discrete-distributions-for","slug":"conjugated-discrete-distributions-for","title":"Conjugated Discrete Distributions for Distributional Reinforcement Learning","date":"2021-12-14","arxiv_id":"2112.07424","repositories_listed":1,"syntology":null},{"url":"/paper/conservative-and-adaptive-penalty-for-model","slug":"conservative-and-adaptive-penalty-for-model","title":"Conservative and Adaptive Penalty for Model-Based Safe Reinforcement Learning","date":"2021-12-14","arxiv_id":"2112.07701","repositories_listed":1,"syntology":null},{"url":"/paper/stochastic-actor-executor-critic-for-image-to","slug":"stochastic-actor-executor-critic-for-image-to","title":"Stochastic Actor-Executor-Critic for Image-to-Image Translation","date":"2021-12-14","arxiv_id":"2112.07403","repositories_listed":1,"syntology":null},{"url":"/paper/stochastic-planner-actor-critic-for","slug":"stochastic-planner-actor-critic-for","title":"Stochastic Planner-Actor-Critic for Unsupervised Deformable Image Registration","date":"2021-12-14","arxiv_id":"2112.07415","repositories_listed":1,"syntology":null},{"url":"/paper/continual-learning-in-environments-with","slug":"continual-learning-in-environments-with","title":"Continual Learning In Environments With Polynomial Mixing Times","date":"2021-12-13","arxiv_id":"2112.07066","repositories_listed":1,"syntology":null},{"url":"/paper/finrl-meta-a-universe-of-near-real-market","slug":"finrl-meta-a-universe-of-near-real-market","title":"FinRL-Meta: A Universe of Near-Real Market Environments for Data-Driven Deep Reinforcement Learning in Quantitative Finance","date":"2021-12-13","arxiv_id":"2112.06753","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/finrl-meta-a-universe-of-near-real-market#ran","syntology_url":"https://syntology.ai/paper/2112.06753","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.06753"}},"official":{"repos":["ai4finance-foundation/finrl-meta"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/human-level-control-through-directly-trained","slug":"human-level-control-through-directly-trained","title":"Human-Level Control through Directly-Trained Deep Spiking Q-Networks","date":"2021-12-13","arxiv_id":"2201.07211","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/human-level-control-through-directly-trained#ran","syntology_url":"https://syntology.ai/paper/2201.07211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2201.07211"}},"official":{"repos":["aptx395/deep-spiking-q-networks"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/pantheonrl-a-marl-library-for-dynamic","slug":"pantheonrl-a-marl-library-for-dynamic","title":"PantheonRL: A MARL Library for Dynamic Training Interactions","date":"2021-12-13","arxiv_id":"2112.07013","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pantheonrl-a-marl-library-for-dynamic#ran","syntology_url":"https://syntology.ai/paper/2112.07013","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.07013"}},"official":{"repos":["Stanford-ILIAD/PantheonRL"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tree-based-focused-web-crawling-with","slug":"tree-based-focused-web-crawling-with","title":"Tree-based Focused Web Crawling with Reinforcement Learning","date":"2021-12-12","arxiv_id":"2112.07620","repositories_listed":1,"syntology":null},{"url":"/paper/elegantrl-podracer-scalable-and-elastic","slug":"elegantrl-podracer-scalable-and-elastic","title":"ElegantRL-Podracer: Scalable and Elastic Library for Cloud-Native Deep Reinforcement Learning","date":"2021-12-11","arxiv_id":"2112.05923","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/elegantrl-podracer-scalable-and-elastic#ran","syntology_url":"https://syntology.ai/paper/2112.05923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.05923"}},"official":{"repos":["ai4finance-foundation/elegantrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ostrichrl-a-musculoskeletal-ostrich","slug":"ostrichrl-a-musculoskeletal-ostrich","title":"OstrichRL: A Musculoskeletal Ostrich Simulation to Study Bio-mechanical Locomotion","date":"2021-12-11","arxiv_id":"2112.06061","repositories_listed":1,"syntology":null},{"url":"/paper/blockwise-sequential-model-learning-for","slug":"blockwise-sequential-model-learning-for","title":"Blockwise Sequential Model Learning for Partially Observable Reinforcement Learning","date":"2021-12-10","arxiv_id":"2112.05343","repositories_listed":1,"syntology":null},{"url":"/paper/deep-q-network-with-proximal-iteration-1","slug":"deep-q-network-with-proximal-iteration-1","title":"Faster Deep Reinforcement Learning with Slower Online Network","date":"2021-12-10","arxiv_id":"2112.05848","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-q-network-with-proximal-iteration-1#ran","syntology_url":"https://syntology.ai/paper/2112.05848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.05848"}},"official":{"repos":["amazon-research/fast-rl-with-slow-updates"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/how-private-is-your-rl-policy-an-inverse-rl","slug":"how-private-is-your-rl-policy-an-inverse-rl","title":"How Private Is Your RL Policy? An Inverse RL Based Analysis Framework","date":"2021-12-10","arxiv_id":"2112.05495","repositories_listed":1,"syntology":null},{"url":"/paper/learning-multiple-gaits-of-quadruped-robot","slug":"learning-multiple-gaits-of-quadruped-robot","title":"Learning multiple gaits of quadruped robot using hierarchical reinforcement learning","date":"2021-12-09","arxiv_id":"2112.04741","repositories_listed":1,"syntology":null},{"url":"/paper/value-function-factorisation-with-hypergraph","slug":"value-function-factorisation-with-hypergraph","title":"Cooperative Multi-Agent Reinforcement Learning with Hypergraph Convolution","date":"2021-12-09","arxiv_id":"2112.06771","repositories_listed":1,"syntology":null},{"url":"/paper/specializing-versatile-skill-libraries-using","slug":"specializing-versatile-skill-libraries-using","title":"Specializing Versatile Skill Libraries using Local Mixture of Experts","date":"2021-12-08","arxiv_id":"2112.04216","repositories_listed":1,"syntology":null},{"url":"/paper/attention-based-model-and-deep-reinforcement","slug":"attention-based-model-and-deep-reinforcement","title":"Attention-Based Model and Deep Reinforcement Learning for Distribution of Event Processing Tasks","date":"2021-12-07","arxiv_id":"2112.03835","repositories_listed":1,"syntology":null},{"url":"/paper/federated-deep-reinforcement-learning-for-the","slug":"federated-deep-reinforcement-learning-for-the","title":"Federated Deep Reinforcement Learning for the Distributed Control of NextG Wireless Networks","date":"2021-12-07","arxiv_id":"2112.03465","repositories_listed":1,"syntology":null},{"url":"/paper/godot-reinforcement-learning-agents","slug":"godot-reinforcement-learning-agents","title":"Godot Reinforcement Learning Agents","date":"2021-12-07","arxiv_id":"2112.03636","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/godot-reinforcement-learning-agents#ran","syntology_url":"https://syntology.ai/paper/2112.03636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.03636"}},"official":{"repos":["edbeeching/godot_rl_agents"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/tell-me-why-explanations-support-learning-of","slug":"tell-me-why-explanations-support-learning-of","title":"Tell me why! Explanations support learning relational and causal structure","date":"2021-12-07","arxiv_id":"2112.03753","repositories_listed":1,"syntology":null},{"url":"/paper/flexible-option-learning-1","slug":"flexible-option-learning-1","title":"Flexible Option Learning","date":"2021-12-06","arxiv_id":"2112.03097","repositories_listed":1,"syntology":null},{"url":"/paper/functional-regularization-for-reinforcement-1","slug":"functional-regularization-for-reinforcement-1","title":"Functional Regularization for Reinforcement Learning via Learned Fourier Features","date":"2021-12-06","arxiv_id":"2112.03257","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/functional-regularization-for-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2112.03257","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.03257"}},"official":{"repos":["alexlioralexli/learned-fourier-features"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-reinforcement-learning-with-5","slug":"hierarchical-reinforcement-learning-with-5","title":"Hierarchical Reinforcement Learning with Timed Subgoals","date":"2021-12-06","arxiv_id":"2112.03100","repositories_listed":1,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/hierarchical-reinforcement-learning-with-5#ran","syntology_url":"https://syntology.ai/paper/2112.03100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.03100"}},"official":{"repos":["martius-lab/hits"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":[]}}},{"url":"/paper/mdpgt-momentum-based-decentralized-policy","slug":"mdpgt-momentum-based-decentralized-policy","title":"MDPGT: Momentum-based Decentralized Policy Gradient Tracking","date":"2021-12-06","arxiv_id":"2112.02813","repositories_listed":1,"syntology":null},{"url":"/paper/offline-pre-trained-multi-agent-decision-1","slug":"offline-pre-trained-multi-agent-decision-1","title":"Offline Pre-trained Multi-Agent Decision Transformer: One Big Sequence Model Tackles All SMAC Tasks","date":"2021-12-06","arxiv_id":"2112.02845","repositories_listed":1,"syntology":null},{"url":"/paper/virtual-replay-cache","slug":"virtual-replay-cache","title":"Virtual Replay Cache","date":"2021-12-06","arxiv_id":"2112.03421","repositories_listed":1,"syntology":null},{"url":"/paper/enhancement-of-a-state-of-the-art-rl-based","slug":"enhancement-of-a-state-of-the-art-rl-based","title":"Enhancement of a state-of-the-art RL-based detection algorithm for Massive MIMO radars","date":"2021-12-05","arxiv_id":"2112.02628","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-automatic-1","slug":"reinforcement-learning-based-automatic-1","title":"Reinforcement Learning-Based Automatic Berthing System","date":"2021-12-03","arxiv_id":"2112.01879","repositories_listed":1,"syntology":null},{"url":"/paper/sample-complexity-of-robust-reinforcement","slug":"sample-complexity-of-robust-reinforcement","title":"Sample Complexity of Robust Reinforcement Learning with a Generative Model","date":"2021-12-02","arxiv_id":"2112.01506","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sample-complexity-of-robust-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2112.01506","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2112.01506"}},"official":{"repos":["kishanpb/RobustRL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sparrl-graph-sparsification-via-deep-1","slug":"sparrl-graph-sparsification-via-deep-1","title":"A Generic Graph Sparsification Framework using Deep Reinforcement Learning","date":"2021-12-02","arxiv_id":"2112.01565","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-data-augmentation-for-1","slug":"automatic-data-augmentation-for-1","title":"Automatic Data Augmentation for Generalization in Reinforcement Learning","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/bcorle-lambda-an-offline-reinforcement","slug":"bcorle-lambda-an-offline-reinforcement","title":"BCORLE($\\lambda$): An Offline Reinforcement Learning and Evaluation Framework for Coupons Allocation in E-commerce Market","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/co-pilot-collaborative-planning-and","slug":"co-pilot-collaborative-planning-and","title":"CO-PILOT: COllaborative Planning and reInforcement Learning On sub-Task curriculum","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/counterexample-guided-rl-policy-refinement","slug":"counterexample-guided-rl-policy-refinement","title":"Counterexample Guided RL Policy Refinement Using Bayesian Optimization","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-domain-adaptation-for-cost","slug":"cross-modal-domain-adaptation-for-cost","title":"Cross-modal Domain Adaptation for Cost-Efficient Visual Reinforcement Learning","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/edge-explaining-deep-reinforcement-learning","slug":"edge-explaining-deep-reinforcement-learning","title":"EDGE: Explaining Deep Reinforcement Learning Policies","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/explicable-reward-design-for-reinforcement","slug":"explicable-reward-design-for-reinforcement","title":"Explicable Reward Design for Reinforcement Learning Agents","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/neorl-neuroevolution-optimization-with","slug":"neorl-neuroevolution-optimization-with","title":"NEORL: NeuroEvolution Optimization with Reinforcement Learning","date":"2021-12-01","arxiv_id":"2112.07057","repositories_listed":1,"syntology":null},{"url":"/paper/offline-model-based-adaptable-policy-learning","slug":"offline-model-based-adaptable-policy-learning","title":"Offline Model-based Adaptable Policy Learning","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/regularized-softmax-deep-multi-agent-q","slug":"regularized-softmax-deep-multi-agent-q","title":"Regularized Softmax Deep Multi-Agent Q-Learning","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-with-state-observation","slug":"reinforcement-learning-with-state-observation","title":"Reinforcement Learning with State Observation Costs in Action-Contingent Noiselessly Observable Markov Decision Processes","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/safe-exploration-for-constrained","slug":"safe-exploration-for-constrained","title":"DOPE: Doubly Optimistic and Pessimistic Exploration for Safe Reinforcement Learning","date":"2021-12-01","arxiv_id":"2112.00885","repositories_listed":1,"syntology":null},{"url":"/paper/symbolic-regression-via-deep-reinforcement","slug":"symbolic-regression-via-deep-reinforcement","title":"Symbolic Regression via Deep Reinforcement Learning Enhanced Genetic Programming Seeding","date":"2021-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/continuous-control-with-ensemble-deep-1","slug":"continuous-control-with-ensemble-deep-1","title":"Continuous Control With Ensemble Deep Deterministic Policy Gradients","date":"2021-11-30","arxiv_id":"2111.15382","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-end-to-end-learning-for-wdm","slug":"model-based-end-to-end-learning-for-wdm","title":"Model-Based End-to-End Learning for WDM Systems With Transceiver Hardware Impairments","date":"2021-11-29","arxiv_id":"2111.14515","repositories_listed":1,"syntology":null},{"url":"/paper/robust-on-policy-data-collection-for-data","slug":"robust-on-policy-data-collection-for-data","title":"Robust On-Policy Sampling for Data-Efficient Policy Evaluation in Reinforcement Learning","date":"2021-11-29","arxiv_id":"2111.14552","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-on-policy-data-collection-for-data#ran","syntology_url":"https://syntology.ai/paper/2111.14552","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2111.14552"}},"official":{"repos":["uoe-agents/robust_onpolicy_data_collection"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-q-learning-based-reinforcement-learning","slug":"deep-q-learning-based-reinforcement-learning","title":"Deep Q-Learning based Reinforcement Learning Approach for Network Intrusion Detection","date":"2021-11-27","arxiv_id":"2111.13978","repositories_listed":1,"syntology":null}],"record_sha256":"b81937b0dacc707f858fcbb8d9302b5b585653346ab1c71a9765343ac372a21c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}