{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/10","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":10,"pages_in_order":59,"rows_per_page":100,"rows":[901,1000],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/9","next":"/task/deep-reinforcement-learning/papers/11","papers":[{"url":"/paper/safe-and-robust-experience-sharing-for","slug":"safe-and-robust-experience-sharing-for","title":"Safe and Robust Experience Sharing for Deterministic Policy Gradient Algorithms","date":"2022-07-27","arxiv_id":"2207.13453","repositories_listed":1,"syntology":null},{"url":"/paper/learning-bipedal-walking-on-planned-footsteps","slug":"learning-bipedal-walking-on-planned-footsteps","title":"Learning Bipedal Walking On Planned Footsteps For Humanoid Robots","date":"2022-07-26","arxiv_id":"2207.12644","repositories_listed":1,"syntology":null},{"url":"/paper/learning-soccer-juggling-skills-with-layer","slug":"learning-soccer-juggling-skills-with-layer","title":"Learning Soccer Juggling Skills with Layer-wise Mixture-of-Experts","date":"2022-07-24","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-energies-of-the","slug":"reinforcement-learning-for-energies-of-the","title":"Reinforcement learning for Energies of the future and carbon neutrality: a Challenge Design","date":"2022-07-21","arxiv_id":"2207.10330","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-market-making-1","slug":"deep-reinforcement-learning-for-market-making-1","title":"Deep Reinforcement Learning for Market Making Under a Hawkes Process-Based Limit Order Book Model","date":"2022-07-20","arxiv_id":"2207.09951","repositories_listed":1,"syntology":null},{"url":"/paper/abstract-demonstrations-and-adaptive","slug":"abstract-demonstrations-and-adaptive","title":"Abstract Demonstrations and Adaptive Exploration for Efficient and Stable Multi-step Sparse Reward Reinforcement Learning","date":"2022-07-19","arxiv_id":"2207.09243","repositories_listed":1,"syntology":null},{"url":"/paper/magpie-automatically-tuning-static-parameters","slug":"magpie-automatically-tuning-static-parameters","title":"Magpie: Automatically Tuning Static Parameters for Distributed File Systems using Deep Reinforcement Learning","date":"2022-07-19","arxiv_id":"2207.09298","repositories_listed":1,"syntology":null},{"url":"/paper/bootstrap-state-representation-using-style","slug":"bootstrap-state-representation-using-style","title":"Bootstrap State Representation using Style Transfer for Better Generalization in Deep Reinforcement Learning","date":"2022-07-15","arxiv_id":"2207.07749","repositories_listed":1,"syntology":null},{"url":"/paper/asset-allocation-from-markowitz-to-deep","slug":"asset-allocation-from-markowitz-to-deep","title":"Asset Allocation: From Markowitz to Deep Reinforcement Learning","date":"2022-07-14","arxiv_id":"2208.07158","repositories_listed":1,"syntology":null},{"url":"/paper/solving-the-traveling-salesperson-problem","slug":"solving-the-traveling-salesperson-problem","title":"Solving the Traveling Salesperson Problem with Precedence Constraints by Deep Reinforcement Learning","date":"2022-07-04","arxiv_id":"2207.01443","repositories_listed":1,"syntology":null},{"url":"/paper/renaissance-robot-optimal-transport-policy","slug":"renaissance-robot-optimal-transport-policy","title":"Renaissance Robot: Optimal Transport Policy Fusion for Learning Diverse Skills","date":"2022-07-03","arxiv_id":"2207.00978","repositories_listed":1,"syntology":null},{"url":"/paper/stabilizing-off-policy-deep-reinforcement","slug":"stabilizing-off-policy-deep-reinforcement","title":"Stabilizing Off-Policy Deep Reinforcement Learning from Pixels","date":"2022-07-03","arxiv_id":"2207.00986","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/stabilizing-off-policy-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2207.00986","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2207.00986"}},"official":{"repos":["aladoro/stabilizing-off-policy-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/interactive-query-assisted-summarization-via","slug":"interactive-query-assisted-summarization-via","title":"Interactive Query-Assisted Summarization via Deep Reinforcement Learning","date":"2022-07-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/colonoscopy-navigation-using-end-to-end-deep","slug":"colonoscopy-navigation-using-end-to-end-deep","title":"Colonoscopy Navigation using End-to-End Deep Visuomotor Control: A User Study","date":"2022-06-30","arxiv_id":"2206.15086","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-swin","slug":"deep-reinforcement-learning-with-swin","title":"Deep Reinforcement Learning with Swin Transformers","date":"2022-06-30","arxiv_id":"2206.15269","repositories_listed":1,"syntology":null},{"url":"/paper/mastering-the-game-of-stratego-with-model","slug":"mastering-the-game-of-stratego-with-model","title":"Mastering the Game of Stratego with Model-Free Multiagent Reinforcement Learning","date":"2022-06-30","arxiv_id":"2206.15378","repositories_listed":1,"syntology":null},{"url":"/paper/conditionally-elicitable-dynamic-risk","slug":"conditionally-elicitable-dynamic-risk","title":"Conditionally Elicitable Dynamic Risk Measures for Deep Reinforcement Learning","date":"2022-06-29","arxiv_id":"2206.14666","repositories_listed":1,"syntology":null},{"url":"/paper/daydreamer-world-models-for-physical-robot","slug":"daydreamer-world-models-for-physical-robot","title":"DayDreamer: World Models for Physical Robot Learning","date":"2022-06-28","arxiv_id":"2206.14176","repositories_listed":1,"syntology":null},{"url":"/paper/improving-policy-optimization-with-generalist","slug":"improving-policy-optimization-with-generalist","title":"Improving Policy Optimization with Generalist-Specialist Learning","date":"2022-06-26","arxiv_id":"2206.12984","repositories_listed":1,"syntology":null},{"url":"/paper/tackling-asymmetric-and-circular-sequential","slug":"tackling-asymmetric-and-circular-sequential","title":"Tackling Asymmetric and Circular Sequential Social Dilemmas with Reinforcement Learning and Graph-based Tit-for-Tat","date":"2022-06-26","arxiv_id":"2206.12909","repositories_listed":1,"syntology":null},{"url":"/paper/toward-multi-target-self-organizing-pursuit","slug":"toward-multi-target-self-organizing-pursuit","title":"Toward multi-target self-organizing pursuit in a partially observable Markov game","date":"2022-06-24","arxiv_id":"2206.12330","repositories_listed":1,"syntology":null},{"url":"/paper/finite-expression-method-for-solving-high","slug":"finite-expression-method-for-solving-high","title":"Finite Expression Method for Solving High-Dimensional Partial Differential Equations","date":"2022-06-21","arxiv_id":"2206.10121","repositories_listed":1,"syntology":null},{"url":"/paper/robust-deep-reinforcement-learning-through-1","slug":"robust-deep-reinforcement-learning-through-1","title":"Robust Deep Reinforcement Learning through Bootstrapped Opportunistic Curriculum","date":"2022-06-21","arxiv_id":"2206.10057","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-through-1#ran","syntology_url":"https://syntology.ai/paper/2206.10057","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.10057"}},"official":{"repos":["jlwu002/bcl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sampling-efficient-deep-reinforcement","slug":"sampling-efficient-deep-reinforcement","title":"Sampling Efficient Deep Reinforcement Learning through Preference-Guided Stochastic Exploration","date":"2022-06-20","arxiv_id":"2206.09627","repositories_listed":1,"syntology":null},{"url":"/paper/an-embedded-feature-selection-framework-for","slug":"an-embedded-feature-selection-framework-for","title":"An Embedded Feature Selection Framework for Control","date":"2022-06-19","arxiv_id":"2206.11064","repositories_listed":1,"syntology":null},{"url":"/paper/smpl-simulated-industrial-manufacturing-and","slug":"smpl-simulated-industrial-manufacturing-and","title":"SMPL: Simulated Industrial Manufacturing and Process Control Learning Environments","date":"2022-06-17","arxiv_id":"2206.08851","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/smpl-simulated-industrial-manufacturing-and#ran","syntology_url":"https://syntology.ai/paper/2206.08851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.08851"}},"official":{"repos":["smpl-env/smpl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-state-of-sparse-training-in-deep","slug":"the-state-of-sparse-training-in-deep","title":"The State of Sparse Training in Deep Reinforcement Learning","date":"2022-06-17","arxiv_id":"2206.10369","repositories_listed":1,"syntology":null},{"url":"/paper/search-based-testing-approach-for-deep","slug":"search-based-testing-approach-for-deep","title":"A Search-Based Testing Approach for Deep Reinforcement Learning Agents","date":"2022-06-15","arxiv_id":"2206.07813","repositories_listed":1,"syntology":null},{"url":"/paper/defending-observation-attacks-in-deep","slug":"defending-observation-attacks-in-deep","title":"Defending Observation Attacks in Deep Reinforcement Learning via Detection and Denoising","date":"2022-06-14","arxiv_id":"2206.07188","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/defending-observation-attacks-in-deep#ran","syntology_url":"https://syntology.ai/paper/2206.07188","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2206.07188"}},"official":{"repos":["ZikangXiong/rl-detect-and-denoise-defense"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rosgas-adaptive-social-bot-detection-with","slug":"rosgas-adaptive-social-bot-detection-with","title":"RoSGAS: Adaptive Social Bot Detection with Reinforced Self-Supervised GNN Architecture Search","date":"2022-06-14","arxiv_id":"2206.06757","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-based-placement-of","slug":"reinforcement-learning-based-placement-of","title":"Reinforcement Learning-based Placement of Charging Stations in Urban Road Networks","date":"2022-06-13","arxiv_id":"2206.06011","repositories_listed":1,"syntology":null},{"url":"/paper/towards-safe-reinforcement-learning-via-2","slug":"towards-safe-reinforcement-learning-via-2","title":"Towards Safe Reinforcement Learning via Constraining Conditional Value-at-Risk","date":"2022-06-09","arxiv_id":"2206.04436","repositories_listed":1,"syntology":null},{"url":"/paper/deeptpi-test-point-insertion-with-deep","slug":"deeptpi-test-point-insertion-with-deep","title":"DeepTPI: Test Point Insertion with Deep Reinforcement Learning","date":"2022-06-07","arxiv_id":"2206.06975","repositories_listed":1,"syntology":null},{"url":"/paper/challenges-to-solving-combinatorially-hard","slug":"challenges-to-solving-combinatorially-hard","title":"Challenges to Solving Combinatorially Hard Long-Horizon Deep RL Tasks","date":"2022-06-03","arxiv_id":"2206.01812","repositories_listed":1,"syntology":null},{"url":"/paper/graph-backup-data-efficient-backup-exploiting","slug":"graph-backup-data-efficient-backup-exploiting","title":"Graph Backup: Data Efficient Backup Exploiting Markovian Transitions","date":"2022-05-31","arxiv_id":"2205.15824","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-reward-poisoning-attacks-on-online","slug":"efficient-reward-poisoning-attacks-on-online","title":"Efficient Reward Poisoning Attacks on Online Deep Reinforcement Learning","date":"2022-05-30","arxiv_id":"2205.14842","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/efficient-reward-poisoning-attacks-on-online#ran","syntology_url":"https://syntology.ai/paper/2205.14842","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.14842"}},"official":{"repos":["yinglunxu/reward_poisoning_attack_drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rlx2-training-a-sparse-deep-reinforcement","slug":"rlx2-training-a-sparse-deep-reinforcement","title":"RLx2: Training a Sparse Deep Reinforcement Learning Model from Scratch","date":"2022-05-30","arxiv_id":"2205.15043","repositories_listed":1,"syntology":null},{"url":"/paper/drlcomplex-reconstruction-of-protein","slug":"drlcomplex-reconstruction-of-protein","title":"DRLComplex: Reconstruction of protein quaternary structures using deep reinforcement learning","date":"2022-05-26","arxiv_id":"2205.13594","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-network-reconfiguration-for-entropy","slug":"dynamic-network-reconfiguration-for-entropy","title":"Dynamic Network Reconfiguration for Entropy Maximization using Deep Reinforcement Learning","date":"2022-05-26","arxiv_id":"2205.13578","repositories_listed":1,"syntology":null},{"url":"/paper/sym-nco-leveraging-symmetricity-for-neural","slug":"sym-nco-leveraging-symmetricity-for-neural","title":"Sym-NCO: Leveraging Symmetricity for Neural Combinatorial Optimization","date":"2022-05-26","arxiv_id":"2205.13209","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/sym-nco-leveraging-symmetricity-for-neural#ran","syntology_url":"https://syntology.ai/paper/2205.13209","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.13209"}},"official":{"repos":["alstn12088/sym-nco"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/treenhance-an-automatic-tree-search-based","slug":"treenhance-an-automatic-tree-search-based","title":"TreEnhance: A Tree Search Method For Low-Light Image Enhancement","date":"2022-05-25","arxiv_id":"2205.12639","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-multi-class","slug":"deep-reinforcement-learning-for-multi-class","title":"Deep Reinforcement Learning for Multi-class Imbalanced Training","date":"2022-05-24","arxiv_id":"2205.12070","repositories_listed":1,"syntology":null},{"url":"/paper/emergent-communication-through-metropolis","slug":"emergent-communication-through-metropolis","title":"Emergent Communication through Metropolis-Hastings Naming Game with Deep Generative Models","date":"2022-05-24","arxiv_id":"2205.12392","repositories_listed":1,"syntology":null},{"url":"/paper/memory-efficient-reinforcement-learning-with","slug":"memory-efficient-reinforcement-learning-with","title":"Memory-efficient Reinforcement Learning with Value-based Knowledge Consolidation","date":"2022-05-22","arxiv_id":"2205.10868","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/memory-efficient-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2205.10868","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.10868"}},"official":{"repos":["qlan3/MeDQN"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/task-relabelling-for-multi-task-transfer","slug":"task-relabelling-for-multi-task-transfer","title":"Task Relabelling for Multi-task Transfer using Successor Features","date":"2022-05-20","arxiv_id":"2205.10175","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-time-1","slug":"deep-reinforcement-learning-for-time-1","title":"Deep Reinforcement Learning for Time Allocation and Directional Transmission in Joint Radar-Communication","date":"2022-05-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/dexterous-robotic-manipulation-using-deep","slug":"dexterous-robotic-manipulation-using-deep","title":"Dexterous Robotic Manipulation using Deep Reinforcement Learning and Knowledge Transfer for Complex Sparse Reward-based Tasks","date":"2022-05-19","arxiv_id":"2205.09683","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dexterous-robotic-manipulation-using-deep#ran","syntology_url":"https://syntology.ai/paper/2205.09683","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2205.09683"}},"official":{"repos":["wq13552463699/rrc_pos_ori"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a2c-is-a-special-case-of-ppo","slug":"a2c-is-a-special-case-of-ppo","title":"A2C is a special case of PPO","date":"2022-05-18","arxiv_id":"2205.09123","repositories_listed":1,"syntology":null},{"url":"/paper/neighborhood-mixup-experience-replay-local","slug":"neighborhood-mixup-experience-replay-local","title":"Neighborhood Mixup Experience Replay: Local Convex Interpolation for Improved Sample Efficiency in Continuous Control Tasks","date":"2022-05-18","arxiv_id":"2205.09117","repositories_listed":1,"syntology":null},{"url":"/paper/the-primacy-bias-in-deep-reinforcement","slug":"the-primacy-bias-in-deep-reinforcement","title":"The Primacy Bias in Deep Reinforcement Learning","date":"2022-05-16","arxiv_id":"2205.07802","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-computational","slug":"deep-reinforcement-learning-for-computational","title":"Deep Reinforcement Learning for Computational Fluid Dynamics on HPC Systems","date":"2022-05-13","arxiv_id":"2205.06502","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-reflecting-surface-configurations","slug":"intelligent-reflecting-surface-configurations","title":"Intelligent Reflecting Surface Configurations for Smart Radio Using Deep Reinforcement Learning","date":"2022-05-11","arxiv_id":"2205.05269","repositories_listed":1,"syntology":null},{"url":"/paper/pervasive-machine-learning-for-smart-radio","slug":"pervasive-machine-learning-for-smart-radio","title":"Pervasive Machine Learning for Smart Radio Environments Enabled by Reconfigurable Intelligent Surfaces","date":"2022-05-08","arxiv_id":"2205.03793","repositories_listed":1,"syntology":null},{"url":"/paper/fire-burns-sword-cuts-commonsense-inductive","slug":"fire-burns-sword-cuts-commonsense-inductive","title":"Fire Burns, Sword Cuts: Commonsense Inductive Bias for Exploration in Text-based Games","date":"2022-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/alphazero-inspired-general-board-game","slug":"alphazero-inspired-general-board-game","title":"AlphaZero-Inspired Game Learning: Faster Training by Using MCTS Only at Test Time","date":"2022-04-28","arxiv_id":"2204.13307","repositories_listed":1,"syntology":null},{"url":"/paper/social-learning-spontaneously-emerges-by","slug":"social-learning-spontaneously-emerges-by","title":"Social learning spontaneously emerges by searching optimal heuristics with deep reinforcement learning","date":"2022-04-26","arxiv_id":"2204.12371","repositories_listed":1,"syntology":null},{"url":"/paper/hypernca-growing-developmental-networks-with","slug":"hypernca-growing-developmental-networks-with","title":"HyperNCA: Growing Developmental Networks with Neural Cellular Automata","date":"2022-04-25","arxiv_id":"2204.11674","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hypernca-growing-developmental-networks-with#ran","syntology_url":"https://syntology.ai/paper/2204.11674","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.11674"}},"official":null}},{"url":"/paper/multi-objective-pointer-network-for","slug":"multi-objective-pointer-network-for","title":"Multi-objective Pointer Network for Combinatorial Optimization","date":"2022-04-25","arxiv_id":"2204.11860","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-objective-pointer-network-for#ran","syntology_url":"https://syntology.ai/paper/2204.11860","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.11860"}},"official":{"repos":["gaoly/mopn"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-a-two-echelon","slug":"deep-reinforcement-learning-for-a-two-echelon","title":"Comparing Deep Reinforcement Learning Algorithms in Two-Echelon Supply Chains","date":"2022-04-20","arxiv_id":"2204.09603","repositories_listed":1,"syntology":null},{"url":"/paper/divide-conquer-imitation-learning","slug":"divide-conquer-imitation-learning","title":"Divide & Conquer Imitation Learning","date":"2022-04-15","arxiv_id":"2204.07404","repositories_listed":1,"syntology":null},{"url":"/paper/accelerated-policy-learning-with-parallel-1","slug":"accelerated-policy-learning-with-parallel-1","title":"Accelerated Policy Learning with Parallel Differentiable Simulation","date":"2022-04-14","arxiv_id":"2204.07137","repositories_listed":1,"syntology":null},{"url":"/paper/gtlo-a-generalized-and-non-linear-multi","slug":"gtlo-a-generalized-and-non-linear-multi","title":"gTLO: A Generalized and Non-linear Multi-Objective Deep Reinforcement Learning Approach","date":"2022-04-11","arxiv_id":"2204.04988","repositories_listed":1,"syntology":null},{"url":"/paper/douzero-improving-doudizhu-ai-by-opponent","slug":"douzero-improving-doudizhu-ai-by-opponent","title":"DouZero+: Improving DouDizhu AI by Opponent Modeling and Coach-guided Learning","date":"2022-04-06","arxiv_id":"2204.02558","repositories_listed":1,"syntology":null},{"url":"/paper/pandr-fast-adaptation-to-new-environments","slug":"pandr-fast-adaptation-to-new-environments","title":"PAnDR: Fast Adaptation to New Environments from Offline Experiences via Decoupling Policy and Environment Representations","date":"2022-04-06","arxiv_id":"2204.02877","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pandr-fast-adaptation-to-new-environments#ran","syntology_url":"https://syntology.ai/paper/2204.02877","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2204.02877"}},"official":null}},{"url":"/paper/automating-reinforcement-learning-with","slug":"automating-reinforcement-learning-with","title":"Automating Reinforcement Learning with Example-based Resets","date":"2022-04-05","arxiv_id":"2204.02041","repositories_listed":1,"syntology":null},{"url":"/paper/learning-pneumatic-non-prehensile","slug":"learning-pneumatic-non-prehensile","title":"Learning Pneumatic Non-Prehensile Manipulation with a Mobile Blower","date":"2022-04-05","arxiv_id":"2204.02390","repositories_listed":1,"syntology":null},{"url":"/paper/an-optical-controlling-environment-and","slug":"an-optical-controlling-environment-and","title":"An Optical Control Environment for Benchmarking Reinforcement Learning Algorithms","date":"2022-03-23","arxiv_id":"2203.12114","repositories_listed":1,"syntology":null},{"url":"/paper/is-vanilla-policy-gradient-overlooked","slug":"is-vanilla-policy-gradient-overlooked","title":"Is Vanilla Policy Gradient Overlooked? Analyzing Deep Reinforcement Learning for Hanabi","date":"2022-03-22","arxiv_id":"2203.11656","repositories_listed":1,"syntology":null},{"url":"/paper/reccover-detecting-causal-confusion-for","slug":"reccover-detecting-causal-confusion-for","title":"ReCCoVER: Detecting Causal Confusion for Explainable Reinforcement Learning","date":"2022-03-21","arxiv_id":"2203.11211","repositories_listed":1,"syntology":null},{"url":"/paper/microracer-a-didactic-environment-for-deep","slug":"microracer-a-didactic-environment-for-deep","title":"MicroRacer: a didactic environment for Deep Reinforcement Learning","date":"2022-03-20","arxiv_id":"2203.10494","repositories_listed":1,"syntology":null},{"url":"/paper/perceiving-the-world-question-guided","slug":"perceiving-the-world-question-guided","title":"Perceiving the World: Question-guided Reinforcement Learning for Text-based Games","date":"2022-03-20","arxiv_id":"2204.09597","repositories_listed":1,"syntology":null},{"url":"/paper/quantum-multi-agent-reinforcement-learning","slug":"quantum-multi-agent-reinforcement-learning","title":"Quantum Multi-Agent Reinforcement Learning via Variational Quantum Circuit Design","date":"2022-03-20","arxiv_id":"2203.10443","repositories_listed":1,"syntology":null},{"url":"/paper/gac-a-deep-reinforcement-learning-model","slug":"gac-a-deep-reinforcement-learning-model","title":"GAC: A Deep Reinforcement Learning Model Toward User Incentivization in Unknown Social Networks","date":"2022-03-17","arxiv_id":"2203.09578","repositories_listed":1,"syntology":null},{"url":"/paper/learning-from-humans-combining-imitation-and","slug":"learning-from-humans-combining-imitation-and","title":"Combining imitation and deep reinforcement learning to accomplish human-level performance on a virtual foraging task","date":"2022-03-11","arxiv_id":"2203.06250","repositories_listed":1,"syntology":null},{"url":"/paper/near-optimal-deep-reinforcement-learning","slug":"near-optimal-deep-reinforcement-learning","title":"Near-optimal Deep Reinforcement Learning Policies from Data for Zone Temperature Control","date":"2022-03-10","arxiv_id":"2203.05434","repositories_listed":1,"syntology":null},{"url":"/paper/multi-objective-reward-generalization","slug":"multi-objective-reward-generalization","title":"Multi-Objective reward generalization: Improving performance of Deep Reinforcement Learning for applications in single-asset trading","date":"2022-03-09","arxiv_id":"2203.04579","repositories_listed":1,"syntology":null},{"url":"/paper/sage-generating-symbolic-goals-for-myopic","slug":"sage-generating-symbolic-goals-for-myopic","title":"SAGE: Generating Symbolic Goals for Myopic Models in Deep Reinforcement Learning","date":"2022-03-09","arxiv_id":"2203.05079","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-entity-1","slug":"deep-reinforcement-learning-for-entity-1","title":"Deep Reinforcement Learning for Entity Alignment","date":"2022-03-07","arxiv_id":"2203.03315","repositories_listed":1,"syntology":{"n":12,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-for-entity-1#ran","syntology_url":"https://syntology.ai/paper/2203.03315","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.03315"}},"official":{"repos":["guolingbing/rlea"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/andes-gym-a-versatile-environment-for-deep","slug":"andes-gym-a-versatile-environment-for-deep","title":"Andes_gym: A Versatile Environment for Deep Reinforcement Learning in Power Systems","date":"2022-03-02","arxiv_id":"2203.01292","repositories_listed":1,"syntology":null},{"url":"/paper/combining-reinforcement-learning-and-optimal","slug":"combining-reinforcement-learning-and-optimal","title":"Combining Reinforcement Learning and Optimal Transport for the Traveling Salesman Problem","date":"2022-03-02","arxiv_id":"2203.00903","repositories_listed":1,"syntology":null},{"url":"/paper/model-free-neural-lyapunov-control-for-safe","slug":"model-free-neural-lyapunov-control-for-safe","title":"Model-free Neural Lyapunov Control for Safe Robot Navigation","date":"2022-03-02","arxiv_id":"2203.01190","repositories_listed":1,"syntology":null},{"url":"/paper/affordance-learning-from-play-for-sample","slug":"affordance-learning-from-play-for-sample","title":"Affordance Learning from Play for Sample-Efficient Policy Learning","date":"2022-03-01","arxiv_id":"2203.00352","repositories_listed":1,"syntology":null},{"url":"/paper/developing-a-chatbot-system-using-deep","slug":"developing-a-chatbot-system-using-deep","title":"Developing a Chatbot system using Deep Learning based for Universities consultancy","date":"2022-02-28","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rl-pgo-reinforcement-learning-based-planar","slug":"rl-pgo-reinforcement-learning-based-planar","title":"RL-PGO: Reinforcement Learning-based Planar Pose-Graph Optimization","date":"2022-02-26","arxiv_id":"2202.13221","repositories_listed":1,"syntology":null},{"url":"/paper/building-a-3-player-mahjong-ai-using-deep","slug":"building-a-3-player-mahjong-ai-using-deep","title":"Building a 3-Player Mahjong AI using Deep Reinforcement Learning","date":"2022-02-25","arxiv_id":"2202.12847","repositories_listed":1,"syntology":null},{"url":"/paper/blockchain-framework-for-artificial","slug":"blockchain-framework-for-artificial","title":"Blockchain Framework for Artificial Intelligence Computation","date":"2022-02-23","arxiv_id":"2202.11264","repositories_listed":1,"syntology":null},{"url":"/paper/using-deep-reinforcement-learning-with","slug":"using-deep-reinforcement-learning-with","title":"Using Deep Reinforcement Learning with Automatic Curriculum Learning for Mapless Navigation in Intralogistics","date":"2022-02-23","arxiv_id":"2202.11512","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparative-study-of-deep-reinforcement","slug":"a-comparative-study-of-deep-reinforcement","title":"A Comparative Study of Deep Reinforcement Learning-based Transferable Energy Management Strategies for Hybrid Electric Vehicles","date":"2022-02-22","arxiv_id":"2202.11514","repositories_listed":1,"syntology":null},{"url":"/paper/shaping-advice-in-deep-reinforcement-learning","slug":"shaping-advice-in-deep-reinforcement-learning","title":"Shaping Advice in Deep Reinforcement Learning","date":"2022-02-19","arxiv_id":"2202.09489","repositories_listed":1,"syntology":null},{"url":"/paper/cadre-a-cascade-deep-reinforcement-learning","slug":"cadre-a-cascade-deep-reinforcement-learning","title":"CADRE: A Cascade Deep Reinforcement Learning Framework for Vision-based Autonomous Urban Driving","date":"2022-02-17","arxiv_id":"2202.08557","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":7,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 7 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cadre-a-cascade-deep-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2202.08557","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.08557"}},"official":{"repos":["BIT-MCS/Cadre"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":7,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/vrl3-a-data-driven-framework-for-visual-deep","slug":"vrl3-a-data-driven-framework-for-visual-deep","title":"VRL3: A Data-Driven Framework for Visual Deep Reinforcement Learning","date":"2022-02-17","arxiv_id":"2202.10324","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":4,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/vrl3-a-data-driven-framework-for-visual-deep#ran","syntology_url":"https://syntology.ai/paper/2202.10324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2202.10324"}},"official":{"repos":["facebookresearch/drqv2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/soft-actor-critic-deep-reinforcement-learning","slug":"soft-actor-critic-deep-reinforcement-learning","title":"Soft Actor-Critic Deep Reinforcement Learning for Fault Tolerant Flight Control","date":"2022-02-16","arxiv_id":"2202.09262","repositories_listed":1,"syntology":null},{"url":"/paper/energy-efficient-parking-analytics-system","slug":"energy-efficient-parking-analytics-system","title":"Energy-Efficient Parking Analytics System using Deep Reinforcement Learning","date":"2022-02-15","arxiv_id":"2202.08973","repositories_listed":1,"syntology":null},{"url":"/paper/exploring-deep-reinforcement-learning","slug":"exploring-deep-reinforcement-learning","title":"Exploring Deep Reinforcement Learning-Assisted Federated Learning for Online Resource Allocation in Privacy-Persevering EdgeIoT","date":"2022-02-15","arxiv_id":"2202.07391","repositories_listed":1,"syntology":null},{"url":"/paper/uncovering-instabilities-in-variational","slug":"uncovering-instabilities-in-variational","title":"Uncovering Instabilities in Variational-Quantum Deep Q-Networks","date":"2022-02-10","arxiv_id":"2202.05195","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-warfarin-dosing-using-deep","slug":"optimizing-warfarin-dosing-using-deep","title":"Optimizing Warfarin Dosing using Deep Reinforcement Learning","date":"2022-02-07","arxiv_id":"2202.03486","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-multi-item","slug":"reinforcement-learning-for-multi-item","title":"Reinforcement learning for multi-item retrieval in the puzzle-based storage system","date":"2022-02-05","arxiv_id":"2202.03424","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-sequential-experimental-design","slug":"optimizing-sequential-experimental-design","title":"Optimizing Sequential Experimental Design with Deep Reinforcement Learning","date":"2022-02-02","arxiv_id":"2202.00821","repositories_listed":1,"syntology":null},{"url":"/paper/accelerating-deep-reinforcement-learning-for","slug":"accelerating-deep-reinforcement-learning-for","title":"Accelerating Deep Reinforcement Learning for Digital Twin Network Optimization with Evolutionary Strategies","date":"2022-02-01","arxiv_id":"2202.00360","repositories_listed":1,"syntology":null},{"url":"/paper/cotv-cooperative-control-for-traffic-light","slug":"cotv-cooperative-control-for-traffic-light","title":"CoTV: Cooperative Control for Traffic Light Signals and Connected Autonomous Vehicles using Deep Reinforcement Learning","date":"2022-01-31","arxiv_id":"2201.13143","repositories_listed":1,"syntology":null}],"record_sha256":"1de5f64e24653c16daedaabad61b2e5f901d2d1ba54634961160f30afe099be3","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}