{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/7","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":7,"pages_in_order":59,"rows_per_page":100,"rows":[601,700],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/6","next":"/task/deep-reinforcement-learning/papers/8","papers":[{"url":"/paper/learning-curricula-in-open-ended-worlds","slug":"learning-curricula-in-open-ended-worlds","title":"Learning Curricula in Open-Ended Worlds","date":"2023-12-03","arxiv_id":"2312.03126","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-curricula-in-open-ended-worlds#ran","syntology_url":"https://syntology.ai/paper/2312.03126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03126"}},"official":{"repos":["facebookresearch/dcd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-decentralized-task-offloading-and","slug":"towards-decentralized-task-offloading-and","title":"Towards Decentralized Task Offloading and Resource Allocation in User-Centric Mobile Edge Computing","date":"2023-12-03","arxiv_id":"2312.01499","repositories_listed":1,"syntology":null},{"url":"/paper/age-based-scheduling-for-mobile-edge","slug":"age-based-scheduling-for-mobile-edge","title":"Age-Based Scheduling for Mobile Edge Computing: A Deep Reinforcement Learning Approach","date":"2023-12-01","arxiv_id":"2312.00279","repositories_listed":1,"syntology":null},{"url":"/paper/gfn-sr-symbolic-regression-with-generative","slug":"gfn-sr-symbolic-regression-with-generative","title":"GFN-SR: Symbolic Regression with Generative Flow Networks","date":"2023-12-01","arxiv_id":"2312.00396","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-zx-diagrams-with-deep","slug":"optimizing-zx-diagrams-with-deep","title":"Optimizing ZX-Diagrams with Deep Reinforcement Learning","date":"2023-11-30","arxiv_id":"2311.18588","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-replaces-supervision-query","slug":"reinforcement-replaces-supervision-query","title":"Reinforcement Replaces Supervision: Query focused Summarization using Deep Reinforcement Learning","date":"2023-11-29","arxiv_id":"2311.17514","repositories_listed":1,"syntology":null},{"url":"/paper/utilizing-explainability-techniques-for","slug":"utilizing-explainability-techniques-for","title":"Utilizing Explainability Techniques for Reinforcement Learning Model Assurance","date":"2023-11-27","arxiv_id":"2311.15838","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/utilizing-explainability-techniques-for#ran","syntology_url":"https://syntology.ai/paper/2311.15838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.15838"}},"official":{"repos":["mitre/arlin"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/margin-trader-a-reinforcement-learning","slug":"margin-trader-a-reinforcement-learning","title":"Margin Trader: A Reinforcement Learning Framework for Portfolio Management with Margin and Constraints","date":"2023-11-25","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-drl-solution-to-help-reduce-the-cost-in","slug":"a-drl-solution-to-help-reduce-the-cost-in","title":"A DRL solution to help reduce the cost in waiting time of securing a traffic light for cyclists","date":"2023-11-23","arxiv_id":"2311.13905","repositories_listed":1,"syntology":null},{"url":"/paper/nav-q-quantum-deep-reinforcement-learning-for","slug":"nav-q-quantum-deep-reinforcement-learning-for","title":"Nav-Q: Quantum Deep Reinforcement Learning for Collision-Free Navigation of Self-Driving Cars","date":"2023-11-20","arxiv_id":"2311.12875","repositories_listed":1,"syntology":null},{"url":"/paper/decentralized-energy-marketplace-via-nfts-and","slug":"decentralized-energy-marketplace-via-nfts-and","title":"Decentralized Energy Marketplace via NFTs and AI-based Agents","date":"2023-11-17","arxiv_id":"2311.10406","repositories_listed":1,"syntology":null},{"url":"/paper/guaranteeing-control-requirements-via-reward","slug":"guaranteeing-control-requirements-via-reward","title":"Guaranteeing Control Requirements via Reward Shaping in Reinforcement Learning","date":"2023-11-16","arxiv_id":"2311.10026","repositories_listed":1,"syntology":null},{"url":"/paper/clipped-objective-policy-gradients-for","slug":"clipped-objective-policy-gradients-for","title":"Clipped-Objective Policy Gradients for Pessimistic Policy Optimization","date":"2023-11-10","arxiv_id":"2311.05846","repositories_listed":1,"syntology":null},{"url":"/paper/uni-o4-unifying-online-and-offline-deep","slug":"uni-o4-unifying-online-and-offline-deep","title":"Uni-O4: Unifying Online and Offline Deep Reinforcement Learning with Multi-Step On-Policy Optimization","date":"2023-11-06","arxiv_id":"2311.03351","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/uni-o4-unifying-online-and-offline-deep#ran","syntology_url":"https://syntology.ai/paper/2311.03351","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.03351"}},"official":{"repos":["Lei-Kun/Uni-O4"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/qoco-a-qoe-oriented-computation-offloading","slug":"qoco-a-qoe-oriented-computation-offloading","title":"QECO: A QoE-Oriented Computation Offloading Algorithm based on Deep Reinforcement Learning for Mobile Edge Computing","date":"2023-11-04","arxiv_id":"2311.02525","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-symbolic-policy-learning-with","slug":"efficient-symbolic-policy-learning-with","title":"Efficient Symbolic Policy Learning with Differentiable Symbolic Expression","date":"2023-11-02","arxiv_id":"2311.02104","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-framework-for-interpretable-and","slug":"hierarchical-framework-for-interpretable-and","title":"Hierarchical Framework for Interpretable and Probabilistic Model-Based Safe Reinforcement Learning","date":"2023-10-28","arxiv_id":"2310.18811","repositories_listed":1,"syntology":null},{"url":"/paper/neuro-inspired-fragmentation-and-recall-to","slug":"neuro-inspired-fragmentation-and-recall-to","title":"Neuro-Inspired Fragmentation and Recall to Overcome Catastrophic Forgetting in Curiosity","date":"2023-10-26","arxiv_id":"2310.17537","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":4,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/neuro-inspired-fragmentation-and-recall-to#ran","syntology_url":"https://syntology.ai/paper/2310.17537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17537"}},"official":{"repos":["fietelab/farcuriosity"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/conditionally-combining-robot-skills-using","slug":"conditionally-combining-robot-skills-using","title":"Conditionally Combining Robot Skills using Large Language Models","date":"2023-10-25","arxiv_id":"2310.17019","repositories_listed":1,"syntology":null},{"url":"/paper/graph-attention-based-deep-reinforcement","slug":"graph-attention-based-deep-reinforcement","title":"Graph Attention-based Deep Reinforcement Learning for solving the Chinese Postman Problem with Load-dependent costs","date":"2023-10-24","arxiv_id":"2310.15516","repositories_listed":1,"syntology":null},{"url":"/paper/safe-navigation-training-autonomous-vehicles","slug":"safe-navigation-training-autonomous-vehicles","title":"Safe Navigation: Training Autonomous Vehicles using Deep Reinforcement Learning in CARLA","date":"2023-10-23","arxiv_id":"2311.10735","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/safe-navigation-training-autonomous-vehicles#ran","syntology_url":"https://syntology.ai/paper/2311.10735","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10735"}},"official":{"repos":["tejas-deo/safe-navigation-training-autonomous-vehicles-using-deep-reinforcement-learning-in-carla"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/neural-multi-objective-combinatorial","slug":"neural-multi-objective-combinatorial","title":"Neural Multi-Objective Combinatorial Optimization with Diversity Enhancement","date":"2023-10-22","arxiv_id":"2310.15195","repositories_listed":1,"syntology":{"n":8,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/neural-multi-objective-combinatorial#ran","syntology_url":"https://syntology.ai/paper/2310.15195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.15195"}},"official":{"repos":["bill-cjb/nhde"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/stabilizing-reinforcement-learning-control-a","slug":"stabilizing-reinforcement-learning-control-a","title":"Stabilizing reinforcement learning control: A modular framework for optimizing over all stable behavior","date":"2023-10-21","arxiv_id":"2310.14098","repositories_listed":1,"syntology":null},{"url":"/paper/enhanced-low-dimensional-sensing-mapless","slug":"enhanced-low-dimensional-sensing-mapless","title":"Enhanced Low-Dimensional Sensing Mapless Navigation of Terrestrial Mobile Robots Using Double Deep Reinforcement Learning Techniques","date":"2023-10-20","arxiv_id":"2310.13809","repositories_listed":1,"syntology":null},{"url":"/paper/rl-x-a-deep-reinforcement-learning-library","slug":"rl-x-a-deep-reinforcement-learning-library","title":"RL-X: A Deep Reinforcement Learning Library (not only) for RoboCup","date":"2023-10-20","arxiv_id":"2310.13396","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-intelligent-2","slug":"deep-reinforcement-learning-based-intelligent-2","title":"Deep Reinforcement Learning-based Intelligent Traffic Signal Controls with Optimized CO2 emissions","date":"2023-10-19","arxiv_id":"2310.13129","repositories_listed":1,"syntology":null},{"url":"/paper/learning-optimal-integration-of-spatial-and","slug":"learning-optimal-integration-of-spatial-and","title":"Learning optimal integration of spatial and temporal information in noisy chemotaxis","date":"2023-10-16","arxiv_id":"2310.10531","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-knowledge-distillation-for","slug":"leveraging-knowledge-distillation-for","title":"Leveraging Knowledge Distillation for Efficient Deep Reinforcement Learning in Resource-Constrained Environments","date":"2023-10-16","arxiv_id":"2310.10170","repositories_listed":1,"syntology":null},{"url":"/paper/a-partially-supervised-reinforcement-learning","slug":"a-partially-supervised-reinforcement-learning","title":"A Partially Supervised Reinforcement Learning Framework for Visual Active Search","date":"2023-10-15","arxiv_id":"2310.09689","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-partially-supervised-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2310.09689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.09689"}},"official":{"repos":["anindyasarkariith/psrl_vas"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/specialized-deep-residual-policy-safe","slug":"specialized-deep-residual-policy-safe","title":"Specialized Deep Residual Policy Safe Reinforcement Learning-Based Controller for Complex and Continuous State-Action Spaces","date":"2023-10-15","arxiv_id":"2310.14788","repositories_listed":1,"syntology":null},{"url":"/paper/learning-rl-policies-for-joint-beamforming","slug":"learning-rl-policies-for-joint-beamforming","title":"Learning RL-Policies for Joint Beamforming Without Exploration: A Batch Constrained Off-Policy Approach","date":"2023-10-12","arxiv_id":"2310.08660","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-uncovers","slug":"deep-reinforcement-learning-uncovers","title":"Deep reinforcement learning uncovers processes for separating azeotropic mixtures without prior knowledge","date":"2023-10-10","arxiv_id":"2310.06415","repositories_listed":1,"syntology":null},{"url":"/paper/hieros-hierarchical-imagination-on-structured","slug":"hieros-hierarchical-imagination-on-structured","title":"Hieros: Hierarchical Imagination on Structured State Space Sequence World Models","date":"2023-10-08","arxiv_id":"2310.05167","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hieros-hierarchical-imagination-on-structured#ran","syntology_url":"https://syntology.ai/paper/2310.05167","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05167"}},"official":{"repos":["snagnar/hieros"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/surgical-gym-a-high-performance-gpu-based","slug":"surgical-gym-a-high-performance-gpu-based","title":"Surgical Gym: A high-performance GPU-based platform for reinforcement learning with surgical robots","date":"2023-10-07","arxiv_id":"2310.04676","repositories_listed":1,"syntology":null},{"url":"/paper/drift-deep-reinforcement-learning-for-1","slug":"drift-deep-reinforcement-learning-for-1","title":"DRIFT: Deep Reinforcement Learning for Intelligent Floating Platforms Trajectories","date":"2023-10-06","arxiv_id":"2310.04266","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-general-reinforcement-learning-1","slug":"discovering-general-reinforcement-learning-1","title":"Discovering General Reinforcement Learning Algorithms with Adversarial Environment Design","date":"2023-10-04","arxiv_id":"2310.02782","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/discovering-general-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2310.02782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02782"}},"official":{"repos":["EmptyJackson/groove"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/differentially-encoded-observation-spaces-for","slug":"differentially-encoded-observation-spaces-for","title":"Differentially Encoded Observation Spaces for Perceptive Reinforcement Learning","date":"2023-10-03","arxiv_id":"2310.01767","repositories_listed":1,"syntology":null},{"url":"/paper/pre-training-with-synthetic-data-helps","slug":"pre-training-with-synthetic-data-helps","title":"Pre-training with Synthetic Data Helps Offline Reinforcement Learning","date":"2023-10-01","arxiv_id":"2310.00771","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pre-training-with-synthetic-data-helps#ran","syntology_url":"https://syntology.ai/paper/2310.00771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00771"}},"official":{"repos":["victor-wang-902/synthetic-pretrain-rl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cleanba-a-reproducible-and-efficient","slug":"cleanba-a-reproducible-and-efficient","title":"Cleanba: A Reproducible and Efficient Distributed Reinforcement Learning Platform","date":"2023-09-29","arxiv_id":"2310.00036","repositories_listed":1,"syntology":{"n":8,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/cleanba-a-reproducible-and-efficient#ran","syntology_url":"https://syntology.ai/paper/2310.00036","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00036"}},"official":{"repos":["vwxyzjn/cleanba"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/memory-gym-partially-observable-challenges-to","slug":"memory-gym-partially-observable-challenges-to","title":"Memory Gym: Towards Endless Tasks to Benchmark Memory Capabilities of Agents","date":"2023-09-29","arxiv_id":"2309.17207","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/memory-gym-partially-observable-challenges-to#ran","syntology_url":"https://syntology.ai/paper/2309.17207","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.17207"}},"official":{"repos":["marcometer/endless-memory-gym"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-terminate-in-object-navigation","slug":"learning-to-terminate-in-object-navigation","title":"Learning to Terminate in Object Navigation","date":"2023-09-28","arxiv_id":"2309.16164","repositories_listed":1,"syntology":null},{"url":"/paper/towards-human-like-rl-taming-non-naturalistic","slug":"towards-human-like-rl-taming-non-naturalistic","title":"Towards Human-Like RL: Taming Non-Naturalistic Behavior in Deep RL via Adaptive Behavioral Costs in 3D Games","date":"2023-09-27","arxiv_id":"2309.15484","repositories_listed":1,"syntology":null},{"url":"/paper/effective-multi-agent-deep-reinforcement","slug":"effective-multi-agent-deep-reinforcement","title":"Effective Multi-Agent Deep Reinforcement Learning Control with Relative Entropy Regularization","date":"2023-09-26","arxiv_id":"2309.14727","repositories_listed":1,"syntology":null},{"url":"/paper/policy-optimization-in-a-noisy-neighborhood-1","slug":"policy-optimization-in-a-noisy-neighborhood-1","title":"Policy Optimization in a Noisy Neighborhood: On Return Landscapes in Continuous Control","date":"2023-09-26","arxiv_id":"2309.14597","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/policy-optimization-in-a-noisy-neighborhood-1#ran","syntology_url":"https://syntology.ai/paper/2309.14597","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.14597"}},"official":{"repos":["nathanrahn/return-landscapes"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/recurrent-hypernetworks-are-surprisingly","slug":"recurrent-hypernetworks-are-surprisingly","title":"Recurrent Hypernetworks are Surprisingly Strong in Meta-RL","date":"2023-09-26","arxiv_id":"2309.14970","repositories_listed":1,"syntology":null},{"url":"/paper/an-ai-chatbot-for-explaining-deep","slug":"an-ai-chatbot-for-explaining-deep","title":"An AI Chatbot for Explaining Deep Reinforcement Learning Decisions of Service-oriented Systems","date":"2023-09-25","arxiv_id":"2309.14391","repositories_listed":1,"syntology":null},{"url":"/paper/deepaco-neural-enhanced-ant-systems-for","slug":"deepaco-neural-enhanced-ant-systems-for","title":"DeepACO: Neural-enhanced Ant Systems for Combinatorial Optimization","date":"2023-09-25","arxiv_id":"2309.14032","repositories_listed":1,"syntology":null},{"url":"/paper/discovering-the-interpretability-performance","slug":"discovering-the-interpretability-performance","title":"Interpretable Decision Tree Search as a Markov Decision Process","date":"2023-09-22","arxiv_id":"2309.12701","repositories_listed":1,"syntology":null},{"url":"/paper/practical-probabilistic-model-based-deep","slug":"practical-probabilistic-model-based-deep","title":"Practical Probabilistic Model-based Deep Reinforcement Learning by Integrating Dropout Uncertainty and Trajectory Sampling","date":"2023-09-20","arxiv_id":"2309.11089","repositories_listed":1,"syntology":null},{"url":"/paper/racing-control-variable-genetic-programming","slug":"racing-control-variable-genetic-programming","title":"Racing Control Variable Genetic Programming for Symbolic Regression","date":"2023-09-13","arxiv_id":"2309.07934","repositories_listed":1,"syntology":null},{"url":"/paper/safe-and-accelerated-deep-reinforcement","slug":"safe-and-accelerated-deep-reinforcement","title":"Safe and Accelerated Deep Reinforcement Learning-based O-RAN Slicing: A Hybrid Transfer Learning Approach","date":"2023-09-13","arxiv_id":"2309.07265","repositories_listed":1,"syntology":null},{"url":"/paper/self-refined-large-language-model-as","slug":"self-refined-large-language-model-as","title":"Self-Refined Large Language Model as Automated Reward Function Designer for Deep Reinforcement Learning in Robotics","date":"2023-09-13","arxiv_id":"2309.06687","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-approach-for-robotic","slug":"a-reinforcement-learning-approach-for-robotic","title":"A Reinforcement Learning Approach for Robotic Unloading from Visual Observations","date":"2023-09-12","arxiv_id":"2309.06621","repositories_listed":1,"syntology":null},{"url":"/paper/avars-alleviating-unexpected-urban-road","slug":"avars-alleviating-unexpected-urban-road","title":"AVARS -- Alleviating Unexpected Urban Road Traffic Congestion using UAVs","date":"2023-09-10","arxiv_id":"2309.04976","repositories_listed":1,"syntology":null},{"url":"/paper/compositional-learning-of-visually-grounded","slug":"compositional-learning-of-visually-grounded","title":"Compositional Learning of Visually-Grounded Concepts Using Reinforcement","date":"2023-09-08","arxiv_id":"2309.04504","repositories_listed":1,"syntology":null},{"url":"/paper/cpu-frequency-scheduling-of-real-time","slug":"cpu-frequency-scheduling-of-real-time","title":"CPU frequency scheduling of real-time applications on embedded devices with temporal encoding-based deep reinforcement learning","date":"2023-09-07","arxiv_id":"2309.03779","repositories_listed":1,"syntology":null},{"url":"/paper/learning-of-generalizable-and-interpretable","slug":"learning-of-generalizable-and-interpretable","title":"Learning of Generalizable and Interpretable Knowledge in Grid-Based Reinforcement Learning Environments","date":"2023-09-07","arxiv_id":"2309.03651","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-from-hierarchical","slug":"deep-reinforcement-learning-from-hierarchical","title":"Deep Reinforcement Learning from Hierarchical Preference Design","date":"2023-09-06","arxiv_id":"2309.02632","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-recharge-uav-coverage-path","slug":"learning-to-recharge-uav-coverage-path","title":"Learning to Recharge: UAV Coverage Path Planning through Deep Reinforcement Learning","date":"2023-09-06","arxiv_id":"2309.03157","repositories_listed":1,"syntology":null},{"url":"/paper/orl-auditor-dataset-auditing-in-offline-deep","slug":"orl-auditor-dataset-auditing-in-offline-deep","title":"ORL-AUDITOR: Dataset Auditing in Offline Deep Reinforcement Learning","date":"2023-09-06","arxiv_id":"2309.03081","repositories_listed":1,"syntology":null},{"url":"/paper/drl-based-trajectory-tracking-for-motion","slug":"drl-based-trajectory-tracking-for-motion","title":"DRL-Based Trajectory Tracking for Motion-Related Modules in Autonomous Driving","date":"2023-08-30","arxiv_id":"2308.15991","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/drl-based-trajectory-tracking-for-motion#ran","syntology_url":"https://syntology.ai/paper/2308.15991","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.15991"}},"official":{"repos":["marmotatzju/drl-based-trajectory-tracking"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/r-3-on-device-real-time-deep-reinforcement","slug":"r-3-on-device-real-time-deep-reinforcement","title":"R^3: On-device Real-Time Deep Reinforcement Learning for Autonomous Robotics","date":"2023-08-29","arxiv_id":"2308.15039","repositories_listed":1,"syntology":null},{"url":"/paper/edge-generation-scheduling-for-dag-tasks","slug":"edge-generation-scheduling-for-dag-tasks","title":"Edge Generation Scheduling for DAG Tasks Using Deep Reinforcement Learning","date":"2023-08-28","arxiv_id":"2308.14647","repositories_listed":1,"syntology":null},{"url":"/paper/learning-visual-tracking-and-reaching-with","slug":"learning-visual-tracking-and-reaching-with","title":"Learning Visual Tracking and Reaching with Deep Reinforcement Learning on a UR10e Robotic Arm","date":"2023-08-28","arxiv_id":"2308.14652","repositories_listed":1,"syntology":null},{"url":"/paper/pretty-darn-good-control-when-are-approximate","slug":"pretty-darn-good-control-when-are-approximate","title":"Pretty darn good control: when are approximate solutions better than approximate models","date":"2023-08-25","arxiv_id":"2308.13654","repositories_listed":1,"syntology":null},{"url":"/paper/an-intentional-forgetting-driven-self-healing","slug":"an-intentional-forgetting-driven-self-healing","title":"An Intentional Forgetting-Driven Self-Healing Method For Deep Reinforcement Learning Systems","date":"2023-08-23","arxiv_id":"2308.12445","repositories_listed":1,"syntology":null},{"url":"/paper/deploying-deep-reinforcement-learning-systems","slug":"deploying-deep-reinforcement-learning-systems","title":"Deploying Deep Reinforcement Learning Systems: A Taxonomy of Challenges","date":"2023-08-23","arxiv_id":"2308.12438","repositories_listed":1,"syntology":null},{"url":"/paper/learning-in-cooperative-multiagent-systems","slug":"learning-in-cooperative-multiagent-systems","title":"Learning in Cooperative Multiagent Systems Using Cognitive and Machine Models","date":"2023-08-18","arxiv_id":"2308.09219","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-identify-critical-states-for","slug":"learning-to-identify-critical-states-for","title":"Learning to Identify Critical States for Reinforcement Learning from Videos","date":"2023-08-15","arxiv_id":"2308.07795","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-to-identify-critical-states-for#ran","syntology_url":"https://syntology.ai/paper/2308.07795","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.07795"}},"official":{"repos":["ai-initiative-kaust/videorlcs"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-for-financial-index","slug":"reinforcement-learning-for-financial-index","title":"Reinforcement Learning for Financial Index Tracking","date":"2023-08-05","arxiv_id":"2308.02820","repositories_listed":1,"syntology":null},{"url":"/paper/job-shop-scheduling-via-deep-reinforcement","slug":"job-shop-scheduling-via-deep-reinforcement","title":"Job Shop Scheduling via Deep Reinforcement Learning: a Sequence to Sequence approach","date":"2023-08-03","arxiv_id":"2308.01797","repositories_listed":1,"syntology":null},{"url":"/paper/smarla-a-safety-monitoring-approach-for-deep","slug":"smarla-a-safety-monitoring-approach-for-deep","title":"SMARLA: A Safety Monitoring Approach for Deep Reinforcement Learning Agents","date":"2023-08-03","arxiv_id":"2308.02594","repositories_listed":1,"syntology":null},{"url":"/paper/drl4route-a-deep-reinforcement-learning","slug":"drl4route-a-deep-reinforcement-learning","title":"DRL4Route: A Deep Reinforcement Learning Framework for Pick-up and Delivery Route Prediction","date":"2023-07-30","arxiv_id":"2307.16246","repositories_listed":1,"syntology":null},{"url":"/paper/improvable-gap-balancing-for-multi-task","slug":"improvable-gap-balancing-for-multi-task","title":"Improvable Gap Balancing for Multi-Task Learning","date":"2023-07-28","arxiv_id":"2307.15429","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improvable-gap-balancing-for-multi-task#ran","syntology_url":"https://syntology.ai/paper/2307.15429","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.15429"}},"official":{"repos":["yanqidai/igb4mtl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/flare-fingerprinting-deep-reinforcement","slug":"flare-fingerprinting-deep-reinforcement","title":"FLARE: Fingerprinting Deep Reinforcement Learning Agents using Universal Adversarial Masks","date":"2023-07-27","arxiv_id":"2307.14751","repositories_listed":1,"syntology":null},{"url":"/paper/machine-learning-powered-pricing-of-the","slug":"machine-learning-powered-pricing-of-the","title":"Machine Learning-powered Pricing of the Multidimensional Passport Option","date":"2023-07-27","arxiv_id":"2307.14887","repositories_listed":1,"syntology":null},{"url":"/paper/a-constraint-enforcement-deep-reinforcement","slug":"a-constraint-enforcement-deep-reinforcement","title":"A Constraint Enforcement Deep Reinforcement Learning Framework for Optimal Energy Storage Systems Dispatch","date":"2023-07-26","arxiv_id":"2307.14304","repositories_listed":1,"syntology":null},{"url":"/paper/2-level-reinforcement-learning-for-ships-on","slug":"2-level-reinforcement-learning-for-ships-on","title":"2-Level Reinforcement Learning for Ships on Inland Waterways: Path Planning and Following","date":"2023-07-25","arxiv_id":"2307.16769","repositories_listed":1,"syntology":null},{"url":"/paper/emergence-of-adaptive-circadian-rhythms-in","slug":"emergence-of-adaptive-circadian-rhythms-in","title":"Emergence of Adaptive Circadian Rhythms in Deep Reinforcement Learning","date":"2023-07-22","arxiv_id":"2307.12143","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emergence-of-adaptive-circadian-rhythms-in#ran","syntology_url":"https://syntology.ai/paper/2307.12143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.12143"}},"official":{"repos":["aqeel13932/mn_project"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/hindsight-dice-stable-credit-assignment-for","slug":"hindsight-dice-stable-credit-assignment-for","title":"Hindsight-DICE: Stable Credit Assignment for Deep Reinforcement Learning","date":"2023-07-21","arxiv_id":"2307.11897","repositories_listed":1,"syntology":null},{"url":"/paper/pomdp-inference-and-robust-solution-via-deep","slug":"pomdp-inference-and-robust-solution-via-deep","title":"POMDP inference and robust solution via deep reinforcement learning: An application to railway optimal maintenance","date":"2023-07-16","arxiv_id":"2307.08082","repositories_listed":1,"syntology":null},{"url":"/paper/aeolus-ocean-a-simulation-environment-for-the","slug":"aeolus-ocean-a-simulation-environment-for-the","title":"Aeolus Ocean -- A simulation environment for the autonomous COLREG-compliant navigation of Unmanned Surface Vehicles using Deep Reinforcement Learning and Maritime Object Detection","date":"2023-07-13","arxiv_id":"2307.06688","repositories_listed":1,"syntology":null},{"url":"/paper/defeating-proactive-jammers-using-deep","slug":"defeating-proactive-jammers-using-deep","title":"Defeating Proactive Jammers Using Deep Reinforcement Learning for Resource-Constrained IoT Networks","date":"2023-07-13","arxiv_id":"2307.06796","repositories_listed":1,"syntology":null},{"url":"/paper/sequential-experimental-design-for-x-ray-ct","slug":"sequential-experimental-design-for-x-ray-ct","title":"Sequential Experimental Design for X-Ray CT Using Deep Reinforcement Learning","date":"2023-07-12","arxiv_id":"2307.06343","repositories_listed":1,"syntology":null},{"url":"/paper/contextual-pre-planning-on-reward-machine","slug":"contextual-pre-planning-on-reward-machine","title":"Contextual Pre-planning on Reward Machine Abstractions for Enhanced Transfer in Deep Reinforcement Learning","date":"2023-07-11","arxiv_id":"2307.05209","repositories_listed":1,"syntology":null},{"url":"/paper/containergym-a-real-world-reinforcement","slug":"containergym-a-real-world-reinforcement","title":"ContainerGym: A Real-World Reinforcement Learning Benchmark for Resource Allocation","date":"2023-07-06","arxiv_id":"2307.02991","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-solve-tasks-with-exploring-prior","slug":"learning-to-solve-tasks-with-exploring-prior","title":"Learning to Solve Tasks with Exploring Prior Behaviours","date":"2023-07-06","arxiv_id":"2307.02889","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-feature-based-deep-reinforcement","slug":"dynamic-feature-based-deep-reinforcement","title":"Dynamic Feature-based Deep Reinforcement Learning for Flow Control of Circular Cylinder with Sparse Surface Pressure Sensing","date":"2023-07-05","arxiv_id":"2307.01995","repositories_listed":1,"syntology":null},{"url":"/paper/learning-symbolic-rules-over-abstract-meaning","slug":"learning-symbolic-rules-over-abstract-meaning","title":"Learning Symbolic Rules over Abstract Meaning Representations for Textual Reinforcement Learning","date":"2023-07-05","arxiv_id":"2307.02689","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-symbolic-rules-over-abstract-meaning#ran","syntology_url":"https://syntology.ai/paper/2307.02689","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.02689"}},"official":{"repos":["ibm/loa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-objective-deep-reinforcement-learning-1","slug":"multi-objective-deep-reinforcement-learning-1","title":"Multi-objective Deep Reinforcement Learning for Mobile Edge Computing","date":"2023-07-05","arxiv_id":"2307.14346","repositories_listed":1,"syntology":null},{"url":"/paper/deep-attention-q-network-for-personalized","slug":"deep-attention-q-network-for-personalized","title":"Deep Attention Q-Network for Personalized Treatment Recommendation","date":"2023-07-04","arxiv_id":"2307.01519","repositories_listed":1,"syntology":null},{"url":"/paper/towards-safe-autonomous-driving-policies","slug":"towards-safe-autonomous-driving-policies","title":"Towards Safe Autonomous Driving Policies using a Neuro-Symbolic Deep Reinforcement Learning Approach","date":"2023-07-03","arxiv_id":"2307.01316","repositories_listed":1,"syntology":null},{"url":"/paper/end-to-end-reinforcement-learning-for-online","slug":"end-to-end-reinforcement-learning-for-online","title":"Learning Coverage Paths in Unknown Environments with Deep Reinforcement Learning","date":"2023-06-29","arxiv_id":"2306.16978","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/end-to-end-reinforcement-learning-for-online#ran","syntology_url":"https://syntology.ai/paper/2306.16978","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.16978"}},"official":{"repos":["arvijj/rl-cpp"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/comprehensive-training-and-evaluation-on-deep","slug":"comprehensive-training-and-evaluation-on-deep","title":"Comprehensive Training and Evaluation on Deep Reinforcement Learning for Automated Driving in Various Simulated Driving Maneuvers","date":"2023-06-20","arxiv_id":"2306.11466","repositories_listed":1,"syntology":null},{"url":"/paper/neural-inventory-control-in-networks-via","slug":"neural-inventory-control-in-networks-via","title":"Neural Inventory Control in Networks via Hindsight Differentiable Policy Optimization","date":"2023-06-20","arxiv_id":"2306.11246","repositories_listed":1,"syntology":null},{"url":"/paper/adaptive-ordered-information-extraction-with","slug":"adaptive-ordered-information-extraction-with","title":"Adaptive Ordered Information Extraction with Deep Reinforcement Learning","date":"2023-06-19","arxiv_id":"2306.10787","repositories_listed":1,"syntology":null},{"url":"/paper/adastop-sequential-testing-for-efficient-and","slug":"adastop-sequential-testing-for-efficient-and","title":"AdaStop: adaptive statistical testing for sound comparisons of Deep RL agents","date":"2023-06-19","arxiv_id":"2306.10882","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-multitask","slug":"deep-reinforcement-learning-with-multitask","title":"Deep Reinforcement Learning with Task-Adaptive Retrieval via Hypernetwork","date":"2023-06-19","arxiv_id":"2306.10698","repositories_listed":1,"syntology":null},{"url":"/paper/joint-path-planning-and-power-allocation-of-a","slug":"joint-path-planning-and-power-allocation-of-a","title":"Joint Path planning and Power Allocation of a Cellular-Connected UAV using Apprenticeship Learning via Deep Inverse Reinforcement Learning","date":"2023-06-15","arxiv_id":"2306.10071","repositories_listed":1,"syntology":null},{"url":"/paper/quadswarm-a-modular-multi-quadrotor-simulator","slug":"quadswarm-a-modular-multi-quadrotor-simulator","title":"QuadSwarm: A Modular Multi-Quadrotor Simulator for Deep Reinforcement Learning with Direct Thrust Control","date":"2023-06-15","arxiv_id":"2306.09537","repositories_listed":1,"syntology":null}],"record_sha256":"c12d902d1348c4b3906473a2d936db8aa3ffe589c6b2ae46e5b3d87917ac863f","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}