{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/8","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":8,"pages_in_order":59,"rows_per_page":100,"rows":[701,800],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/7","next":"/task/deep-reinforcement-learning/papers/9","papers":[{"url":"/paper/ocatari-object-centric-atari-2600","slug":"ocatari-object-centric-atari-2600","title":"OCAtari: Object-Centric Atari 2600 Reinforcement Learning Environments","date":"2023-06-14","arxiv_id":"2306.08649","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ocatari-object-centric-atari-2600#ran","syntology_url":"https://syntology.ai/paper/2306.08649","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.08649"}},"official":{"repos":["k4ntz/oc_atari"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/denselight-efficient-control-for-large-scale","slug":"denselight-efficient-control-for-large-scale","title":"DenseLight: Efficient Control for Large-scale Traffic Signals with Dense Feedback","date":"2023-06-13","arxiv_id":"2306.07553","repositories_listed":1,"syntology":null},{"url":"/paper/learned-spatial-data-partitioning","slug":"learned-spatial-data-partitioning","title":"Learned spatial data partitioning","date":"2023-06-08","arxiv_id":"2306.04846","repositories_listed":1,"syntology":null},{"url":"/paper/agent-performing-autonomous-stock-trading","slug":"agent-performing-autonomous-stock-trading","title":"Agent Performing Autonomous Stock Trading under Good and Bad Situations","date":"2023-06-06","arxiv_id":"2306.03985","repositories_listed":1,"syntology":null},{"url":"/paper/backproptools-a-fast-portable-deep","slug":"backproptools-a-fast-portable-deep","title":"RLtools: A Fast, Portable Deep Reinforcement Learning Library for Continuous Control","date":"2023-06-06","arxiv_id":"2306.03530","repositories_listed":1,"syntology":null},{"url":"/paper/meta-sage-scale-meta-learning-scheduled","slug":"meta-sage-scale-meta-learning-scheduled","title":"Meta-SAGE: Scale Meta-Learning Scheduled Adaptation with Guided Exploration for Mitigating Scale Shift on Combinatorial Optimization","date":"2023-06-05","arxiv_id":"2306.02688","repositories_listed":1,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":7,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/meta-sage-scale-meta-learning-scheduled#ran","syntology_url":"https://syntology.ai/paper/2306.02688","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.02688"}},"official":{"repos":["kaist-silab/meta-sage"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/seizing-serendipity-exploiting-the-value-of","slug":"seizing-serendipity-exploiting-the-value-of","title":"Seizing Serendipity: Exploiting the Value of Past Success in Off-Policy Actor-Critic","date":"2023-06-05","arxiv_id":"2306.02865","repositories_listed":1,"syntology":null},{"url":"/paper/average-aoi-minimization-for-energy","slug":"average-aoi-minimization-for-energy","title":"Average AoI Minimization for Energy Harvesting Relay-aided Status Update Network Using Deep Reinforcement Learning","date":"2023-06-02","arxiv_id":"2306.01251","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-framework-for-1","slug":"deep-reinforcement-learning-framework-for-1","title":"Deep Reinforcement Learning Framework for Thoracic Diseases Classification via Prior Knowledge Guidance","date":"2023-06-02","arxiv_id":"2306.01232","repositories_listed":1,"syntology":null},{"url":"/paper/hyperparameters-in-reinforcement-learning-and","slug":"hyperparameters-in-reinforcement-learning-and","title":"Hyperparameters in Reinforcement Learning and How To Tune Them","date":"2023-06-02","arxiv_id":"2306.01324","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hyperparameters-in-reinforcement-learning-and#ran","syntology_url":"https://syntology.ai/paper/2306.01324","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.01324"}},"official":{"repos":["facebookresearch/how-to-autorl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/relu-to-the-rescue-improve-your-on-policy","slug":"relu-to-the-rescue-improve-your-on-policy","title":"ReLU to the Rescue: Improve Your On-Policy Actor-Critic with Positive Advantages","date":"2023-06-02","arxiv_id":"2306.01460","repositories_listed":1,"syntology":null},{"url":"/paper/symmetric-exploration-in-combinatorial","slug":"symmetric-exploration-in-combinatorial","title":"Symmetric Replay Training: Enhancing Sample Efficiency in Deep Reinforcement Learning for Combinatorial Optimization","date":"2023-06-02","arxiv_id":"2306.01276","repositories_listed":1,"syntology":null},{"url":"/paper/fast-matrix-multiplication-without-tears-a","slug":"fast-matrix-multiplication-without-tears-a","title":"Fast Matrix Multiplication Without Tears: A Constraint Programming Approach","date":"2023-06-01","arxiv_id":"2306.01097","repositories_listed":1,"syntology":null},{"url":"/paper/dhrl-fnmr-an-intelligent-multicast-routing","slug":"dhrl-fnmr-an-intelligent-multicast-routing","title":"DHRL-FNMR: An Intelligent Multicast Routing Approach Based on Deep Hierarchical Reinforcement Learning in SDN","date":"2023-05-30","arxiv_id":"2305.19077","repositories_listed":1,"syntology":null},{"url":"/paper/a-hierarchical-approach-to-population","slug":"a-hierarchical-approach-to-population","title":"A Hierarchical Approach to Population Training for Human-AI Collaboration","date":"2023-05-26","arxiv_id":"2305.16708","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-pd-control-using-deep-reinforcement","slug":"adaptive-pd-control-using-deep-reinforcement","title":"Adaptive PD Control using Deep Reinforcement Learning for Local-Remote Teleoperation with Stochastic Time Delays","date":"2023-05-26","arxiv_id":"2305.16979","repositories_listed":1,"syntology":null},{"url":"/paper/physical-deep-reinforcement-learning-safety","slug":"physical-deep-reinforcement-learning-safety","title":"Physics-Regulated Deep Reinforcement Learning: Invariant Embeddings","date":"2023-05-26","arxiv_id":"2305.16614","repositories_listed":1,"syntology":null},{"url":"/paper/counterfactual-explainer-framework-for-deep","slug":"counterfactual-explainer-framework-for-deep","title":"Counterfactual Explainer Framework for Deep Reinforcement Learning Models Using Policy Distillation","date":"2023-05-25","arxiv_id":"2305.16532","repositories_listed":1,"syntology":null},{"url":"/paper/market-making-with-deep-reinforcement","slug":"market-making-with-deep-reinforcement","title":"Market Making with Deep Reinforcement Learning from Limit Order Books","date":"2023-05-25","arxiv_id":"2305.15821","repositories_listed":1,"syntology":null},{"url":"/paper/road-planning-for-slums-via-deep","slug":"road-planning-for-slums-via-deep","title":"Road Planning for Slums via Deep Reinforcement Learning","date":"2023-05-22","arxiv_id":"2305.13060","repositories_listed":1,"syntology":null},{"url":"/paper/testing-of-deep-reinforcement-learning-agents","slug":"testing-of-deep-reinforcement-learning-agents","title":"Testing of Deep Reinforcement Learning Agents with Surrogate Models","date":"2023-05-22","arxiv_id":"2305.12751","repositories_listed":1,"syntology":null},{"url":"/paper/vision-based-drl-autonomous-driving-agent","slug":"vision-based-drl-autonomous-driving-agent","title":"Vision-based DRL Autonomous Driving Agent with Sim2Real Transfer","date":"2023-05-19","arxiv_id":"2305.11589","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-exploration","slug":"deep-reinforcement-learning-based-exploration","title":"Deep Reinforcement Learning-based Exploration of Web Applications","date":"2023-05-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/what-matters-in-reinforcement-learning-for","slug":"what-matters-in-reinforcement-learning-for","title":"What Matters in Reinforcement Learning for Tractography","date":"2023-05-15","arxiv_id":"2305.09041","repositories_listed":1,"syntology":null},{"url":"/paper/an-intelligent-sdwn-routing-algorithm-based","slug":"an-intelligent-sdwn-routing-algorithm-based","title":"An Intelligent SDWN Routing Algorithm Based on Network Situational Awareness and Deep Reinforcement Learning","date":"2023-05-12","arxiv_id":"2305.10441","repositories_listed":1,"syntology":null},{"url":"/paper/quantile-based-deep-reinforcement-learning","slug":"quantile-based-deep-reinforcement-learning","title":"Quantile-Based Deep Reinforcement Learning using Two-Timescale Policy Gradient Algorithms","date":"2023-05-12","arxiv_id":"2305.07248","repositories_listed":1,"syntology":null},{"url":"/paper/extracting-diagnosis-pathways-from-electronic","slug":"extracting-diagnosis-pathways-from-electronic","title":"Extracting Diagnosis Pathways from Electronic Health Records Using Deep Reinforcement Learning","date":"2023-05-10","arxiv_id":"2305.06295","repositories_listed":1,"syntology":null},{"url":"/paper/flexible-job-shop-scheduling-via-dual","slug":"flexible-job-shop-scheduling-via-dual","title":"Flexible Job Shop Scheduling via Dual Attention Network Based Reinforcement Learning","date":"2023-05-09","arxiv_id":"2305.05119","repositories_listed":1,"syntology":null},{"url":"/paper/optimal-energy-system-scheduling-using-a","slug":"optimal-energy-system-scheduling-using-a","title":"Optimal Energy System Scheduling Using A Constraint-Aware Reinforcement Learning Algorithm","date":"2023-05-09","arxiv_id":"2305.05484","repositories_listed":1,"syntology":null},{"url":"/paper/train-a-real-world-local-path-planner-in-one","slug":"train-a-real-world-local-path-planner-in-one","title":"Train a Real-world Local Path Planner in One Hour via Partially Decoupled Reinforcement Learning and Vectorized Diversity","date":"2023-05-07","arxiv_id":"2305.04180","repositories_listed":1,"syntology":null},{"url":"/paper/map-based-experience-replay-a-memory","slug":"map-based-experience-replay-a-memory","title":"Map-based Experience Replay: A Memory-Efficient Solution to Catastrophic Forgetting in Reinforcement Learning","date":"2023-05-03","arxiv_id":"2305.02054","repositories_listed":1,"syntology":null},{"url":"/paper/online-portfolio-management-via-deep","slug":"online-portfolio-management-via-deep","title":"Online Portfolio Management via Deep Reinforcement Learning with High-Frequency Data","date":"2023-05-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/posterior-sampling-for-deep-reinforcement","slug":"posterior-sampling-for-deep-reinforcement","title":"Posterior Sampling for Deep Reinforcement Learning","date":"2023-04-30","arxiv_id":"2305.00477","repositories_listed":1,"syntology":null},{"url":"/paper/semi-infinitely-constrained-markov-decision","slug":"semi-infinitely-constrained-markov-decision","title":"Semi-Infinitely Constrained Markov Decision Processes and Efficient Reinforcement Learning","date":"2023-04-29","arxiv_id":"2305.00254","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-object-centric-generalized-value","slug":"discovering-object-centric-generalized-value","title":"Discovering Object-Centric Generalized Value Functions From Pixels","date":"2023-04-27","arxiv_id":"2304.13892","repositories_listed":1,"syntology":{"n":18,"n_ran":14,"n_constructed":0,"n_ran_checked":10,"n_instrument":4,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 4 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/discovering-object-centric-generalized-value#ran","syntology_url":"https://syntology.ai/paper/2304.13892","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.13892"}},"official":{"repos":["somjit77/oc_gvfs"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/socnavgym-a-reinforcement-learning-gym-for","slug":"socnavgym-a-reinforcement-learning-gym-for","title":"SocNavGym: A Reinforcement Learning Gym for Social Navigation","date":"2023-04-27","arxiv_id":"2304.14102","repositories_listed":1,"syntology":null},{"url":"/paper/a-multi-task-approach-to-robust-deep","slug":"a-multi-task-approach-to-robust-deep","title":"A Multi-Task Approach to Robust Deep Reinforcement Learning for Resource Allocation","date":"2023-04-25","arxiv_id":"2304.12660","repositories_listed":1,"syntology":null},{"url":"/paper/proto-value-networks-scaling-representation","slug":"proto-value-networks-scaling-representation","title":"Proto-Value Networks: Scaling Representation Learning with Auxiliary Tasks","date":"2023-04-25","arxiv_id":"2304.12567","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-control-hydrodynamic-force-on-fluidic","slug":"how-to-control-hydrodynamic-force-on-fluidic","title":"How to Control Hydrodynamic Force on Fluidic Pinball via Deep Reinforcement Learning","date":"2023-04-23","arxiv_id":"2304.11526","repositories_listed":1,"syntology":null},{"url":"/paper/robust-deep-reinforcement-learning-scheduling","slug":"robust-deep-reinforcement-learning-scheduling","title":"Robust Deep Reinforcement Learning Scheduling via Weight Anchoring","date":"2023-04-20","arxiv_id":"2304.10176","repositories_listed":1,"syntology":null},{"url":"/paper/temporl-laser-pulse-temporal-shape","slug":"temporl-laser-pulse-temporal-shape","title":"TempoRL: laser pulse temporal shape optimization with Deep Reinforcement Learning","date":"2023-04-20","arxiv_id":"2304.12187","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-rl-theory-and-practice-with-the-1","slug":"bridging-rl-theory-and-practice-with-the-1","title":"Bridging RL Theory and Practice with the Effective Horizon","date":"2023-04-19","arxiv_id":"2304.09853","repositories_listed":1,"syntology":null},{"url":"/paper/h-tsp-hierarchically-solving-the-large-scale","slug":"h-tsp-hierarchically-solving-the-large-scale","title":"H-TSP: Hierarchically Solving the Large-Scale Travelling Salesman Problem","date":"2023-04-19","arxiv_id":"2304.09395","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/h-tsp-hierarchically-solving-the-large-scale#ran","syntology_url":"https://syntology.ai/paper/2304.09395","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.09395"}},"official":{"repos":["Learning4Optimization-HUST/H-TSP"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-policies-for-resource-allocation-in","slug":"learning-policies-for-resource-allocation-in","title":"Learning policies for resource allocation in business processes","date":"2023-04-19","arxiv_id":"2304.09970","repositories_listed":1,"syntology":null},{"url":"/paper/learning-representative-trajectories-of","slug":"learning-representative-trajectories-of","title":"Learning Representative Trajectories of Dynamical Systems via Domain-Adaptive Imitation","date":"2023-04-19","arxiv_id":"2304.10260","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-actor-critic-deep-reinforcement","slug":"benchmarking-actor-critic-deep-reinforcement","title":"Benchmarking Actor-Critic Deep Reinforcement Learning Algorithms for Robotics Control with Action Constraints","date":"2023-04-18","arxiv_id":"2304.08743","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/benchmarking-actor-critic-deep-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2304.08743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.08743"}},"official":{"repos":["omron-sinicx/action-constrained-rl-benchmark"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/collaborative-multi-bs-power-management-for","slug":"collaborative-multi-bs-power-management-for","title":"Collaborative Multi-BS Power Management for Dense Radio Access Network using Deep Reinforcement Learning","date":"2023-04-17","arxiv_id":"2304.07976","repositories_listed":1,"syntology":null},{"url":"/paper/on-fast-converged-reinforcement-learning-for","slug":"on-fast-converged-reinforcement-learning-for","title":"Fast-Converged Deep Reinforcement Learning for Optimal Dispatch of Large-Scale Power Systems under Transient Security Constraints","date":"2023-04-17","arxiv_id":"2304.08320","repositories_listed":1,"syntology":null},{"url":"/paper/a-platform-agnostic-deep-reinforcement","slug":"a-platform-agnostic-deep-reinforcement","title":"A Platform-Agnostic Deep Reinforcement Learning Framework for Effective Sim2Real Transfer towards Autonomous Driving","date":"2023-04-14","arxiv_id":"2304.08235","repositories_listed":1,"syntology":null},{"url":"/paper/eagle-end-to-end-deep-reinforcement-learning","slug":"eagle-end-to-end-deep-reinforcement-learning","title":"Eagle: End-to-end Deep Reinforcement Learning based Autonomous Control of PTZ Cameras","date":"2023-04-10","arxiv_id":"2304.04356","repositories_listed":1,"syntology":null},{"url":"/paper/robopianist-a-benchmark-for-high-dimensional","slug":"robopianist-a-benchmark-for-high-dimensional","title":"RoboPianist: Dexterous Piano Playing with Deep Reinforcement Learning","date":"2023-04-09","arxiv_id":"2304.04150","repositories_listed":1,"syntology":null},{"url":"/paper/generating-a-graph-colouring-heuristic-with","slug":"generating-a-graph-colouring-heuristic-with","title":"Generating a Graph Colouring Heuristic with Deep Q-Learning and Graph Neural Networks","date":"2023-04-08","arxiv_id":"2304.04051","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-mapless","slug":"deep-reinforcement-learning-based-mapless","title":"Deep Reinforcement Learning-Based Mapless Crowd Navigation with Perceived Risk of the Moving Crowd for Mobile Robots","date":"2023-04-07","arxiv_id":"2304.03593","repositories_listed":1,"syntology":null},{"url":"/paper/uav-obstacle-avoidance-by-human-in-the-loop","slug":"uav-obstacle-avoidance-by-human-in-the-loop","title":"UAV Obstacle Avoidance by Human-in-the-Loop Reinforcement in Arbitrary 3D Environment","date":"2023-04-07","arxiv_id":"2304.05959","repositories_listed":1,"syntology":null},{"url":"/paper/effective-control-of-two-dimensional-rayleigh","slug":"effective-control-of-two-dimensional-rayleigh","title":"Effective control of two-dimensional Rayleigh--Bénard convection: invariant multi-agent reinforcement learning is all you need","date":"2023-04-05","arxiv_id":"2304.02370","repositories_listed":1,"syntology":null},{"url":"/paper/humanlight-incentivizing-ridesharing-via","slug":"humanlight-incentivizing-ridesharing-via","title":"HumanLight: Incentivizing Ridesharing via Human-centric Deep Reinforcement Learning in Traffic Signal Control","date":"2023-04-05","arxiv_id":"2304.03697","repositories_listed":1,"syntology":null},{"url":"/paper/optimization-of-binding-affinities-in","slug":"optimization-of-binding-affinities-in","title":"Optimization of binding affinities in chemical space with generative pretrained transformer and deep reinforcement learning","date":"2023-04-03","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/solving-dynamic-traveling-salesman-problems","slug":"solving-dynamic-traveling-salesman-problems","title":"Solving Dynamic Traveling Salesman Problems With Deep Reinforcement Learning","date":"2023-04-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/rlor-a-flexible-framework-of-deep","slug":"rlor-a-flexible-framework-of-deep","title":"RLOR: A Flexible Framework of Deep Reinforcement Learning for Operation Research","date":"2023-03-23","arxiv_id":"2303.13117","repositories_listed":1,"syntology":null},{"url":"/paper/distributed-two-tier-drl-framework-for-cell","slug":"distributed-two-tier-drl-framework-for-cell","title":"Distributed Two-tier DRL Framework for Cell-Free Network: Association, Beamforming and Power Allocation","date":"2023-03-22","arxiv_id":"2303.12479","repositories_listed":1,"syntology":null},{"url":"/paper/wasserstein-auto-encoded-mdps-formal","slug":"wasserstein-auto-encoded-mdps-formal","title":"Wasserstein Auto-encoded MDPs: Formal Verification of Efficiently Distilled RL Policies with Many-sided Guarantees","date":"2023-03-22","arxiv_id":"2303.12558","repositories_listed":1,"syntology":null},{"url":"/paper/conversational-tree-search-a-new-hybrid","slug":"conversational-tree-search-a-new-hybrid","title":"Conversational Tree Search: A New Hybrid Dialog Task","date":"2023-03-17","arxiv_id":"2303.10227","repositories_listed":1,"syntology":null},{"url":"/paper/learning-model-free-robust-precoding-for","slug":"learning-model-free-robust-precoding-for","title":"Learning Model-Free Robust Precoding for Cooperative Multibeam Satellite Communications","date":"2023-03-13","arxiv_id":"2303.11427","repositories_listed":1,"syntology":null},{"url":"/paper/synthetic-experience-replay-1","slug":"synthetic-experience-replay-1","title":"Synthetic Experience Replay","date":"2023-03-12","arxiv_id":"2303.06614","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/synthetic-experience-replay-1#ran","syntology_url":"https://syntology.ai/paper/2303.06614","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.06614"}},"official":{"repos":["conglu1997/SynthER"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/towards-practical-multi-robot-hybrid-tasks","slug":"towards-practical-multi-robot-hybrid-tasks","title":"Towards Practical Multi-Robot Hybrid Tasks Allocation for Autonomous Cleaning","date":"2023-03-12","arxiv_id":"2303.06531","repositories_listed":1,"syntology":null},{"url":"/paper/solving-routing-problems-for-multiple","slug":"solving-routing-problems-for-multiple","title":"Solving routing problems for multiple cooperative Unmanned Aerial Vehicles using Transformer networks, vol. 122, pp. 106085, 2023","date":"2023-03-09","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-bipedal-walking-for-humanoids-with","slug":"learning-bipedal-walking-for-humanoids-with","title":"Learning Bipedal Walking for Humanoids with Current Feedback","date":"2023-03-07","arxiv_id":"2303.03724","repositories_listed":1,"syntology":null},{"url":"/paper/map-elites-with-descriptor-conditioned","slug":"map-elites-with-descriptor-conditioned","title":"MAP-Elites with Descriptor-Conditioned Gradients and Archive Distillation into a Single Policy","date":"2023-03-07","arxiv_id":"2303.03832","repositories_listed":1,"syntology":null},{"url":"/paper/mastering-strategy-card-game-legends-of-code","slug":"mastering-strategy-card-game-legends-of-code","title":"Mastering Strategy Card Game (Legends of Code and Magic) via End-to-End Policy and Optimistic Smooth Fictitious Play","date":"2023-03-07","arxiv_id":"2303.04096","repositories_listed":1,"syntology":null},{"url":"/paper/deep-symbolic-regression-for-physics-guided","slug":"deep-symbolic-regression-for-physics-guided","title":"Deep symbolic regression for physics guided by units constraints: toward the automated discovery of physical laws","date":"2023-03-06","arxiv_id":"2303.03192","repositories_listed":1,"syntology":null},{"url":"/paper/swim-a-general-purpose-high-performing-and","slug":"swim-a-general-purpose-high-performing-and","title":"Swim: A General-Purpose, High-Performing, and Efficient Activation Function for Locomotion Control Tasks","date":"2023-03-05","arxiv_id":"2303.02640","repositories_listed":1,"syntology":null},{"url":"/paper/ai-generated-incentive-mechanism-and-full","slug":"ai-generated-incentive-mechanism-and-full","title":"AI-Generated Incentive Mechanism and Full-Duplex Semantic Communications for Information Sharing","date":"2023-03-03","arxiv_id":"2303.01896","repositories_listed":1,"syntology":null},{"url":"/paper/decision-transformer-under-random-frame","slug":"decision-transformer-under-random-frame","title":"Decision Transformer under Random Frame Dropping","date":"2023-03-03","arxiv_id":"2303.03391","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":3,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decision-transformer-under-random-frame#ran","syntology_url":"https://syntology.ai/paper/2303.03391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.03391"}},"official":{"repos":["hukz18/defog"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crystalbox-future-based-explanations-for-drl","slug":"crystalbox-future-based-explanations-for-drl","title":"CrystalBox: Future-Based Explanations for Input-Driven Deep RL Systems","date":"2023-02-27","arxiv_id":"2302.13483","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-needs-driven-agent-learning","slug":"hierarchical-needs-driven-agent-learning","title":"Hierarchical Needs-driven Agent Learning Systems: From Deep Reinforcement Learning To Diverse Strategies","date":"2023-02-25","arxiv_id":"2302.13132","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-uncertainty-in-value-functions","slug":"model-based-uncertainty-in-value-functions","title":"Model-Based Uncertainty in Value Functions","date":"2023-02-24","arxiv_id":"2302.12526","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/model-based-uncertainty-in-value-functions#ran","syntology_url":"https://syntology.ai/paper/2302.12526","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.12526"}},"official":{"repos":["boschresearch/ube-mbrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/energy-harvesting-reconfigurable-intelligent","slug":"energy-harvesting-reconfigurable-intelligent","title":"Energy Harvesting Reconfigurable Intelligent Surface for UAV Based on Robust Deep Reinforcement Learning","date":"2023-02-23","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/assessment-of-reinforcement-learning-for","slug":"assessment-of-reinforcement-learning-for","title":"Assessment of Reinforcement Learning for Macro Placement","date":"2023-02-21","arxiv_id":"2302.11014","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-cost","slug":"deep-reinforcement-learning-for-cost","title":"Deep Reinforcement Learning for Cost-Effective Medical Diagnosis","date":"2023-02-20","arxiv_id":"2302.10261","repositories_listed":1,"syntology":null},{"url":"/paper/robust-deep-reinforcement-learning-through-2","slug":"robust-deep-reinforcement-learning-through-2","title":"Regret-Based Defense in Adversarial Reinforcement Learning","date":"2023-02-14","arxiv_id":"2302.06912","repositories_listed":1,"syntology":{"n":11,"n_ran":10,"n_constructed":0,"n_ran_checked":6,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-through-2#ran","syntology_url":"https://syntology.ai/paper/2302.06912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06912"}},"official":{"repos":["romanbelaire/robust-ccer"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-noise-filtering-with-dynamic-sparse","slug":"automatic-noise-filtering-with-dynamic-sparse","title":"Automatic Noise Filtering with Dynamic Sparse Training in Deep Reinforcement Learning","date":"2023-02-13","arxiv_id":"2302.06548","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/automatic-noise-filtering-with-dynamic-sparse#ran","syntology_url":"https://syntology.ai/paper/2302.06548","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06548"}},"official":{"repos":["bramgrooten/automatic-noise-filtering"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uncoupled-learning-of-differential","slug":"uncoupled-learning-of-differential","title":"Uncoupled Learning of Differential Stackelberg Equilibria with Commitments","date":"2023-02-07","arxiv_id":"2302.03438","repositories_listed":1,"syntology":null},{"url":"/paper/intrinsic-rewards-from-self-organizing","slug":"intrinsic-rewards-from-self-organizing","title":"Intrinsic Rewards from Self-Organizing Feature Maps for Exploration in Reinforcement Learning","date":"2023-02-06","arxiv_id":"2302.04125","repositories_listed":1,"syntology":null},{"url":"/paper/model-free-quantum-gate-design-and","slug":"model-free-quantum-gate-design-and","title":"Model-free Quantum Gate Design and Calibration using Deep Reinforcement Learning","date":"2023-02-05","arxiv_id":"2302.02371","repositories_listed":1,"syntology":null},{"url":"/paper/sample-dropout-a-simple-yet-effective","slug":"sample-dropout-a-simple-yet-effective","title":"Sample Dropout: A Simple yet Effective Variance Reduction Technique in Deep Policy Optimization","date":"2023-02-05","arxiv_id":"2302.02299","repositories_listed":1,"syntology":{"n":12,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":12,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/sample-dropout-a-simple-yet-effective#ran","syntology_url":"https://syntology.ai/paper/2302.02299","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.02299"}},"official":{"repos":["linzichuan/sdpo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/execution-based-code-generation-using-deep","slug":"execution-based-code-generation-using-deep","title":"Execution-based Code Generation using Deep Reinforcement Learning","date":"2023-01-31","arxiv_id":"2301.13816","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/execution-based-code-generation-using-deep#ran","syntology_url":"https://syntology.ai/paper/2301.13816","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2301.13816"}},"official":{"repos":["reddy-lab-code-research/PPOCoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/optimal-transport-perturbations-for-safe","slug":"optimal-transport-perturbations-for-safe","title":"Optimal Transport Perturbations for Safe Reinforcement Learning with Robustness Guarantees","date":"2023-01-31","arxiv_id":"2301.13375","repositories_listed":1,"syntology":null},{"url":"/paper/automatic-intrinsic-reward-shaping-for","slug":"automatic-intrinsic-reward-shaping-for","title":"Automatic Intrinsic Reward Shaping for Exploration in Deep Reinforcement Learning","date":"2023-01-26","arxiv_id":"2301.10886","repositories_listed":1,"syntology":null},{"url":"/paper/nesig-a-neuro-symbolic-method-for-learning-to","slug":"nesig-a-neuro-symbolic-method-for-learning-to","title":"NeSIG: A Neuro-Symbolic Method for Learning to Generate Planning Problems","date":"2023-01-24","arxiv_id":"2301.10280","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-path","slug":"deep-reinforcement-learning-based-path","title":"Deep-Reinforcement-Learning-based Path Planning for Industrial Robots using Distance Sensors as Observation","date":"2023-01-14","arxiv_id":"2301.05980","repositories_listed":1,"syntology":null},{"url":"/paper/mutation-testing-of-deep-reinforcement","slug":"mutation-testing-of-deep-reinforcement","title":"Mutation Testing of Deep Reinforcement Learning Based on Real Faults","date":"2023-01-13","arxiv_id":"2301.05651","repositories_listed":1,"syntology":null},{"url":"/paper/schlably-a-python-framework-for-deep","slug":"schlably-a-python-framework-for-deep","title":"schlably: A Python Framework for Deep Reinforcement Learning Based Scheduling Experiments","date":"2023-01-10","arxiv_id":"2301.04182","repositories_listed":1,"syntology":null},{"url":"/paper/why-people-skip-music-on-predicting-music","slug":"why-people-skip-music-on-predicting-music","title":"Why People Skip Music? On Predicting Music Skips using Deep Reinforcement Learning","date":"2023-01-10","arxiv_id":"2301.03881","repositories_listed":1,"syntology":null},{"url":"/paper/centralized-cooperative-exploration-policy","slug":"centralized-cooperative-exploration-policy","title":"Centralized Cooperative Exploration Policy for Continuous Control Tasks","date":"2023-01-06","arxiv_id":"2301.02375","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-irrigation","slug":"deep-reinforcement-learning-for-irrigation","title":"Deep reinforcement learning for irrigation scheduling using high-dimensional sensor feedback","date":"2023-01-02","arxiv_id":"2301.00899","repositories_listed":1,"syntology":null},{"url":"/paper/environment-agnostic-representation-for","slug":"environment-agnostic-representation-for","title":"Environment Agnostic Representation for Visual Reinforcement Learning","date":"2023-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/goal-guided-transformer-enabled-reinforcement","slug":"goal-guided-transformer-enabled-reinforcement","title":"Goal-Guided Transformer-Enabled Reinforcement Learning for Efficient Autonomous Navigation","date":"2023-01-01","arxiv_id":"2301.00362","repositories_listed":1,"syntology":null},{"url":"/paper/transformer-in-transformer-as-backbone-for","slug":"transformer-in-transformer-as-backbone-for","title":"Transformer in Transformer as Backbone for Deep Reinforcement Learning","date":"2022-12-30","arxiv_id":"2212.14538","repositories_listed":1,"syntology":null},{"url":"/paper/tuning-synaptic-connections-instead-of","slug":"tuning-synaptic-connections-instead-of","title":"Tuning Synaptic Connections instead of Weights by Genetic Algorithm in Spiking Policy Network","date":"2022-12-29","arxiv_id":"2301.10292","repositories_listed":1,"syntology":null},{"url":"/paper/on-pathologies-in-kl-regularized-1","slug":"on-pathologies-in-kl-regularized-1","title":"On Pathologies in KL-Regularized Reinforcement Learning from Expert Demonstrations","date":"2022-12-28","arxiv_id":"2212.13936","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/on-pathologies-in-kl-regularized-1#ran","syntology_url":"https://syntology.ai/paper/2212.13936","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.13936"}},"official":{"repos":["conglu1997/nppac"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"5a74cf3aa67d0d2fd04c3ac241993f171fbb558f3b3bcdc3f2568688e616aa8d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}