{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/13","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":13,"pages_in_order":59,"rows_per_page":100,"rows":[1201,1300],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/12","next":"/task/deep-reinforcement-learning/papers/14","papers":[{"url":"/paper/deepfreight-a-model-free-deep-reinforcement","slug":"deepfreight-a-model-free-deep-reinforcement","title":"DeepFreight: Integrating Deep Reinforcement Learning and Mixed Integer Programming for Multi-transfer Truck Freight Delivery","date":"2021-03-05","arxiv_id":"2103.03450","repositories_listed":1,"syntology":null},{"url":"/paper/improving-computational-efficiency-in-visual","slug":"improving-computational-efficiency-in-visual","title":"Improving Computational Efficiency in Visual Reinforcement Learning via Stored Embeddings","date":"2021-03-04","arxiv_id":"2103.02886","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-urllc-data","slug":"deep-reinforcement-learning-for-urllc-data","title":"Deep Reinforcement Learning for URLLC data management on top of scheduled eMBB traffic","date":"2021-03-02","arxiv_id":"2103.01801","repositories_listed":1,"syntology":null},{"url":"/paper/robot-navigation-in-a-crowd-by-integrating","slug":"robot-navigation-in-a-crowd-by-integrating","title":"Robot Navigation in a Crowd by Integrating Deep Reinforcement Learning and Online Planning","date":"2021-02-26","arxiv_id":"2102.13265","repositories_listed":1,"syntology":null},{"url":"/paper/robust-deep-reinforcement-learning-via-multi","slug":"robust-deep-reinforcement-learning-via-multi","title":"DRIBO: Robust Deep Reinforcement Learning via Multi-View Information Bottleneck","date":"2021-02-26","arxiv_id":"2102.13268","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/robust-deep-reinforcement-learning-via-multi#ran","syntology_url":"https://syntology.ai/paper/2102.13268","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.13268"}},"official":{"repos":["BU-DEPEND-Lab/DRIBO"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/task-agnostic-morphology-evolution-1","slug":"task-agnostic-morphology-evolution-1","title":"Task-Agnostic Morphology Evolution","date":"2021-02-25","arxiv_id":"2102.13100","repositories_listed":1,"syntology":null},{"url":"/paper/memory-based-deep-reinforcement-learning-for-1","slug":"memory-based-deep-reinforcement-learning-for-1","title":"Memory-based Deep Reinforcement Learning for POMDPs","date":"2021-02-24","arxiv_id":"2102.12344","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/memory-based-deep-reinforcement-learning-for-1#ran","syntology_url":"https://syntology.ai/paper/2102.12344","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.12344"}},"official":{"repos":["LinghengMeng/LSTM-TD3"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/modular-deep-reinforcement-learning-for","slug":"modular-deep-reinforcement-learning-for","title":"Modular Deep Reinforcement Learning for Continuous Motion Planning with Temporal Logic","date":"2021-02-24","arxiv_id":"2102.12855","repositories_listed":1,"syntology":null},{"url":"/paper/accelerated-sim-to-real-deep-reinforcement","slug":"accelerated-sim-to-real-deep-reinforcement","title":"Accelerated Sim-to-Real Deep Reinforcement Learning: Learning Collision Avoidance from Human Player","date":"2021-02-21","arxiv_id":"2102.10711","repositories_listed":1,"syntology":null},{"url":"/paper/causal-inference-q-network-toward-resilient-1","slug":"causal-inference-q-network-toward-resilient-1","title":"Training a Resilient Q-Network against Observational Interference","date":"2021-02-18","arxiv_id":"2102.09677","repositories_listed":1,"syntology":null},{"url":"/paper/intelligent-electric-vehicle-charging","slug":"intelligent-electric-vehicle-charging","title":"Intelligent Electric Vehicle Charging Recommendation Based on Multi-Agent Reinforcement Learning","date":"2021-02-15","arxiv_id":"2102.07359","repositories_listed":1,"syntology":null},{"url":"/paper/scaling-multi-agent-reinforcement-learning","slug":"scaling-multi-agent-reinforcement-learning","title":"Scaling Multi-Agent Reinforcement Learning with Selective Parameter Sharing","date":"2021-02-15","arxiv_id":"2102.07475","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/scaling-multi-agent-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2102.07475","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.07475"}},"official":{"repos":["uoe-agents/seps"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/scrofazero-mastering-trick-taking-poker-game","slug":"scrofazero-mastering-trick-taking-poker-game","title":"ScrofaZero: Mastering Trick-taking Poker Game Gongzhu by Deep Reinforcement Learning","date":"2021-02-15","arxiv_id":"2102.07495","repositories_listed":1,"syntology":null},{"url":"/paper/ltl2action-generalizing-ltl-instructions-for","slug":"ltl2action-generalizing-ltl-instructions-for","title":"LTL2Action: Generalizing LTL Instructions for Multi-Task RL","date":"2021-02-13","arxiv_id":"2102.06858","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ltl2action-generalizing-ltl-instructions-for#ran","syntology_url":"https://syntology.ai/paper/2102.06858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.06858"}},"official":{"repos":["LTL2Action/LTL2Action"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/q-value-weighted-regression-reinforcement-1","slug":"q-value-weighted-regression-reinforcement-1","title":"Q-Value Weighted Regression: Reinforcement Learning with Limited Data","date":"2021-02-12","arxiv_id":"2102.06782","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/q-value-weighted-regression-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2102.06782","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.06782"}},"official":null}},{"url":"/paper/deep-reinforcement-agent-for-scheduling-in","slug":"deep-reinforcement-agent-for-scheduling-in","title":"Deep Reinforcement Agent for Scheduling in HPC","date":"2021-02-11","arxiv_id":"2102.06243","repositories_listed":1,"syntology":null},{"url":"/paper/domain-adaptation-in-reinforcement-learning","slug":"domain-adaptation-in-reinforcement-learning","title":"Domain Adaptation In Reinforcement Learning Via Latent Unified State Representation","date":"2021-02-10","arxiv_id":"2102.05714","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":3,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/domain-adaptation-in-reinforcement-learning#ran","syntology_url":"https://syntology.ai/paper/2102.05714","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2102.05714"}},"official":{"repos":["KarlXing/LUSR"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/policy-augmentation-an-exploration-strategy","slug":"policy-augmentation-an-exploration-strategy","title":"Policy Augmentation: An Exploration Strategy for Faster Convergence of Deep Reinforcement Learning Algorithms","date":"2021-02-10","arxiv_id":"2102.05249","repositories_listed":1,"syntology":null},{"url":"/paper/adversarially-guided-actor-critic-1","slug":"adversarially-guided-actor-critic-1","title":"Adversarially Guided Actor-Critic","date":"2021-02-08","arxiv_id":"2102.04376","repositories_listed":1,"syntology":null},{"url":"/paper/rl-scope-cross-stack-profiling-for-deep","slug":"rl-scope-cross-stack-profiling-for-deep","title":"RL-Scope: Cross-Stack Profiling for Deep Reinforcement Learning Workloads","date":"2021-02-08","arxiv_id":"2102.04285","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-reinforcement-learning-for","slug":"explainable-reinforcement-learning-for","title":"Explainable Reinforcement Learning for Longitudinal Control","date":"2021-02-06","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-smart","slug":"deep-reinforcement-learning-for-smart","title":"Deep reinforcement learning for smart calibration of radio telescopes","date":"2021-02-05","arxiv_id":"2102.03200","repositories_listed":1,"syntology":null},{"url":"/paper/proactive-and-aoi-aware-failure-recovery-for","slug":"proactive-and-aoi-aware-failure-recovery-for","title":"Proactive and AoI-aware Failure Recovery for Stateful NFV-enabled Zero-Touch 6G Networks: Model-Free DRL Approach","date":"2021-02-02","arxiv_id":"2103.03817","repositories_listed":1,"syntology":null},{"url":"/paper/gymd2d-a-device-to-device-underlay-cellular","slug":"gymd2d-a-device-to-device-underlay-cellular","title":"GymD2D: A Device-to-Device Underlay Cellular Offload Evaluation Platform","date":"2021-01-27","arxiv_id":"2101.11188","repositories_listed":1,"syntology":null},{"url":"/paper/differentiable-trust-region-layers-for-deep-1","slug":"differentiable-trust-region-layers-for-deep-1","title":"Differentiable Trust Region Layers for Deep Reinforcement Learning","date":"2021-01-22","arxiv_id":"2101.09207","repositories_listed":1,"syntology":null},{"url":"/paper/theory-of-mind-for-deep-reinforcement","slug":"theory-of-mind-for-deep-reinforcement","title":"Theory of Mind for Deep Reinforcement Learning in Hanabi","date":"2021-01-22","arxiv_id":"2101.09328","repositories_listed":1,"syntology":null},{"url":"/paper/unifying-cardiovascular-modelling-with-deep","slug":"unifying-cardiovascular-modelling-with-deep","title":"Unifying Cardiovascular Modelling with Deep Reinforcement Learning for Uncertainty Aware Control of Sepsis Treatment","date":"2021-01-21","arxiv_id":"2101.08477","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/unifying-cardiovascular-modelling-with-deep#ran","syntology_url":"https://syntology.ai/paper/2101.08477","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.08477"}},"official":{"repos":["thxsxth/POMDP_RLSepsis"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/mt5b3-a-framework-for-building","slug":"mt5b3-a-framework-for-building","title":"mt5se: An Open Source Framework for Building Autonomous Trading Robots","date":"2021-01-20","arxiv_id":"2101.08169","repositories_listed":1,"syntology":null},{"url":"/paper/shielding-atari-games-with-bounded-prescience","slug":"shielding-atari-games-with-bounded-prescience","title":"Shielding Atari Games with Bounded Prescience","date":"2021-01-20","arxiv_id":"2101.08153","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-producing","slug":"deep-reinforcement-learning-for-producing","title":"Deep Reinforcement Learning for Producing Furniture Layout in Indoor Scenes","date":"2021-01-19","arxiv_id":"2101.07462","repositories_listed":1,"syntology":null},{"url":"/paper/towards-facilitating-empathic-conversations","slug":"towards-facilitating-empathic-conversations","title":"Towards Facilitating Empathic Conversations in Online Mental Health Support: A Reinforcement Learning Approach","date":"2021-01-19","arxiv_id":"2101.07714","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-perturbation-based-saliency-maps","slug":"benchmarking-perturbation-based-saliency-maps","title":"Benchmarking Perturbation-based Saliency Maps for Explaining Atari Agents","date":"2021-01-18","arxiv_id":"2101.07312","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-active-high","slug":"deep-reinforcement-learning-for-active-high","title":"Deep Reinforcement Learning for Active High Frequency Trading","date":"2021-01-18","arxiv_id":"2101.07107","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-soccer-player-from-live-camera-to","slug":"evaluating-soccer-player-from-live-camera-to","title":"Evaluating Soccer Player: from Live Camera to Deep Reinforcement Learning","date":"2021-01-13","arxiv_id":"2101.05388","repositories_listed":1,"syntology":null},{"url":"/paper/developing-an-openai-gym-compatible-framework","slug":"developing-an-openai-gym-compatible-framework","title":"Developing an OpenAI Gym-compatible framework and simulation environment for testing Deep Reinforcement Learning agents solving the Ambulance Location Problem","date":"2021-01-12","arxiv_id":"2101.04434","repositories_listed":1,"syntology":null},{"url":"/paper/cross-modal-contrastive-learning-of","slug":"cross-modal-contrastive-learning-of","title":"Cross-Modal Contrastive Learning of Representations for Navigation using Lightweight, Low-Cost Millimeter Wave Radar for Adverse Environmental Conditions","date":"2021-01-10","arxiv_id":"2101.03525","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-function","slug":"deep-reinforcement-learning-with-function","title":"Deep Reinforcement Learning with Function Properties in Mean Reversion Strategies","date":"2021-01-09","arxiv_id":"2101.03418","repositories_listed":1,"syntology":null},{"url":"/paper/a-reinforcement-learning-based-encoder","slug":"a-reinforcement-learning-based-encoder","title":"A Reinforcement Learning Based Encoder-Decoder Framework for Learning Stock Trading Rules","date":"2021-01-08","arxiv_id":"2101.03867","repositories_listed":1,"syntology":null},{"url":"/paper/joint-deep-reinforcement-learning-and","slug":"joint-deep-reinforcement-learning-and","title":"Joint Deep Reinforcement Learning and Unfolding: Beam Selection and Precoding for mmWave Multiuser MIMO with Lens Arrays","date":"2021-01-05","arxiv_id":"2101.01336","repositories_listed":1,"syntology":null},{"url":"/paper/a-novel-policy-for-pre-trained-deep","slug":"a-novel-policy-for-pre-trained-deep","title":"A novel policy for pre-trained Deep Reinforcement Learning for Speech Emotion Recognition","date":"2021-01-04","arxiv_id":"2101.00738","repositories_listed":1,"syntology":null},{"url":"/paper/faults-in-deep-reinforcement-learning","slug":"faults-in-deep-reinforcement-learning","title":"Faults in Deep Reinforcement Learning Programs: A Taxonomy and A Detection Approach","date":"2021-01-01","arxiv_id":"2101.00135","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-meta-reinforcement-learning-for","slug":"hierarchical-meta-reinforcement-learning-for","title":"Hierarchical Meta Reinforcement Learning for Multi-Task Environments","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/regularization-matters-in-policy-optimization","slug":"regularization-matters-in-policy-optimization","title":"Regularization Matters in Policy Optimization - An Empirical Study on Continuous Control","date":"2021-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-portfolio","slug":"deep-reinforcement-learning-for-portfolio","title":"Deep Reinforcement Learning for Long-Short Portfolio Optimization","date":"2020-12-26","arxiv_id":"2012.13773","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-joint-1","slug":"deep-reinforcement-learning-for-joint-1","title":"Deep Reinforcement Learning for Joint Spectrum and Power Allocation in Cellular Networks","date":"2020-12-19","arxiv_id":"2012.10682","repositories_listed":1,"syntology":null},{"url":"/paper/multi-decoder-attention-model-with-embedding","slug":"multi-decoder-attention-model-with-embedding","title":"Multi-Decoder Attention Model with Embedding Glimpse for Solving Vehicle Routing Problems","date":"2020-12-19","arxiv_id":"2012.10638","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-decoder-attention-model-with-embedding#ran","syntology_url":"https://syntology.ai/paper/2012.10638","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.10638"}},"official":{"repos":["liangxinedu/MDAM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/high-throughput-synchronous-deep-rl-1","slug":"high-throughput-synchronous-deep-rl-1","title":"High-Throughput Synchronous Deep RL","date":"2020-12-17","arxiv_id":"2012.09849","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/high-throughput-synchronous-deep-rl-1#ran","syntology_url":"https://syntology.ai/paper/2012.09849","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.09849"}},"official":{"repos":["IouJenLiu/HTS-RL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/model-free-and-bayesian-ensembling-model","slug":"model-free-and-bayesian-ensembling-model","title":"Model-free and Bayesian Ensembling Model-based Deep Reinforcement Learning for Particle Accelerator Control Demonstrated on the FERMI FEL","date":"2020-12-17","arxiv_id":"2012.09737","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-for-contact-rich-tasks","slug":"reinforcement-learning-for-contact-rich-tasks","title":"Reinforcement Learning for Contact-Rich Tasks: Robotic Peg Insertion Strategies","date":"2020-12-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/super-reinforcement-bros-playing-super-mario","slug":"super-reinforcement-bros-playing-super-mario","title":"Super Reinforcement Bros: Playing Super Mario Bros with Reinforcement Learning","date":"2020-12-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/an-efficient-asynchronous-method-for-1","slug":"an-efficient-asynchronous-method-for-1","title":"An Efficient Asynchronous Method for Integrating Evolutionary and Gradient-based Policy Search","date":"2020-12-10","arxiv_id":"2012.05417","repositories_listed":1,"syntology":null},{"url":"/paper/resolving-implicit-coordination-in-multi","slug":"resolving-implicit-coordination-in-multi","title":"Resolving Implicit Coordination in Multi-Agent Deep Reinforcement Learning with Deep Q-Networks & Game Theory","date":"2020-12-08","arxiv_id":"2012.09136","repositories_listed":1,"syntology":null},{"url":"/paper/intelligence-and-learning-in-o-ran-for-data","slug":"intelligence-and-learning-in-o-ran-for-data","title":"Intelligence and Learning in O-RAN for Data-driven NextG Cellular Networks","date":"2020-12-02","arxiv_id":"2012.01263","repositories_listed":1,"syntology":null},{"url":"/paper/learning-multi-agent-communication-through","slug":"learning-multi-agent-communication-through","title":"Learning Multi-Agent Communication through Structured Attentive Reasoning","date":"2020-12-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/continuous-transition-improving-sample","slug":"continuous-transition-improving-sample","title":"Continuous Transition: Improving Sample Efficiency for Continuous Control Problems via MixUp","date":"2020-11-30","arxiv_id":"2011.14487","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-information-diffusion-in-time","slug":"efficient-information-diffusion-in-time","title":"Efficient Information Diffusion in Time-Varying Graphs through Deep Reinforcement Learning","date":"2020-11-27","arxiv_id":"2011.13518","repositories_listed":1,"syntology":null},{"url":"/paper/an-end-to-end-deep-reinforcement-learning","slug":"an-end-to-end-deep-reinforcement-learning","title":"An End-to-end Deep Reinforcement Learning Approach for the Long-term Short-term Planning on the Frenet Space","date":"2020-11-26","arxiv_id":"2011.13098","repositories_listed":1,"syntology":null},{"url":"/paper/combining-semantic-guidance-and-deep","slug":"combining-semantic-guidance-and-deep","title":"Combining Semantic Guidance and Deep Reinforcement Learning For Generating Human Level Paintings","date":"2020-11-25","arxiv_id":"2011.12589","repositories_listed":1,"syntology":null},{"url":"/paper/symmetry-aware-actor-critic-for-3d-molecular-1","slug":"symmetry-aware-actor-critic-for-3d-molecular-1","title":"Symmetry-Aware Actor-Critic for 3D Molecular Design","date":"2020-11-25","arxiv_id":"2011.12747","repositories_listed":1,"syntology":null},{"url":"/paper/world-model-as-a-graph-learning-latent","slug":"world-model-as-a-graph-learning-latent","title":"World Model as a Graph: Learning Latent Landmarks for Planning","date":"2020-11-25","arxiv_id":"2011.12491","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-feedback","slug":"deep-reinforcement-learning-for-feedback","title":"Deep reinforcement learning for feedback control in a collective flashing ratchet","date":"2020-11-20","arxiv_id":"2011.10357","repositories_listed":1,"syntology":null},{"url":"/paper/tfpnp-tuning-free-plug-and-play-proximal","slug":"tfpnp-tuning-free-plug-and-play-proximal","title":"TFPnP: Tuning-free Plug-and-Play Proximal Algorithm with Applications to Inverse Imaging Problems","date":"2020-11-18","arxiv_id":"2012.05703","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/tfpnp-tuning-free-plug-and-play-proximal#ran","syntology_url":"https://syntology.ai/paper/2012.05703","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2012.05703"}},"official":{"repos":["Vandermode/TFPnP"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cdt-cascading-decision-trees-for-explainable-1","slug":"cdt-cascading-decision-trees-for-explainable-1","title":"CDT: Cascading Decision Trees for Explainable Reinforcement Learning","date":"2020-11-15","arxiv_id":"2011.07553","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-cybersecurity","slug":"deep-reinforcement-learning-for-cybersecurity","title":"Deep Reinforcement Learning for Cybersecurity Assessment of Wind Integrated Power Systems","date":"2020-11-15","arxiv_id":"2007.03025","repositories_listed":1,"syntology":null},{"url":"/paper/tonic-a-deep-reinforcement-learning-library","slug":"tonic-a-deep-reinforcement-learning-library","title":"Tonic: A Deep Reinforcement Learning Library for Fast Prototyping and Benchmarking","date":"2020-11-15","arxiv_id":"2011.07537","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/tonic-a-deep-reinforcement-learning-library#ran","syntology_url":"https://syntology.ai/paper/2011.07537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.07537"}},"official":{"repos":["fabiopardo/tonic"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/deepmind-lab2d","slug":"deepmind-lab2d","title":"DeepMind Lab2D","date":"2020-11-13","arxiv_id":"2011.07027","repositories_listed":1,"syntology":null},{"url":"/paper/query-based-targeted-action-space-adversarial","slug":"query-based-targeted-action-space-adversarial","title":"Query-based Targeted Action-Space Adversarial Policies on Deep Reinforcement Learning Agents","date":"2020-11-13","arxiv_id":"2011.07114","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-large-scale-fleet-management-on-a","slug":"optimizing-large-scale-fleet-management-on-a","title":"Optimizing Large-Scale Fleet Management on a Road Network using Multi-Agent Deep Reinforcement Learning with Graph Neural Network","date":"2020-11-12","arxiv_id":"2011.06175","repositories_listed":1,"syntology":null},{"url":"/paper/decentralized-motion-planning-for-multi-robot","slug":"decentralized-motion-planning-for-multi-robot","title":"Decentralized Motion Planning for Multi-Robot Navigation using Deep Reinforcement Learning","date":"2020-11-11","arxiv_id":"2011.05605","repositories_listed":1,"syntology":null},{"url":"/paper/geometric-deep-reinforcement-learning-for","slug":"geometric-deep-reinforcement-learning-for","title":"Geometric Deep Reinforcement Learning for Dynamic DAG Scheduling","date":"2020-11-09","arxiv_id":"2011.04333","repositories_listed":1,"syntology":null},{"url":"/paper/drafting-in-collectible-card-games-via","slug":"drafting-in-collectible-card-games-via","title":"Drafting in Collectible Card Games via Reinforcement Learning","date":"2020-11-07","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/learning-trajectories-for-visual-inertial","slug":"learning-trajectories-for-visual-inertial","title":"Learning Trajectories for Visual-Inertial System Calibration via Model-based Heuristic Deep Reinforcement Learning","date":"2020-11-04","arxiv_id":"2011.02574","repositories_listed":1,"syntology":null},{"url":"/paper/amortized-variational-deep-q-network","slug":"amortized-variational-deep-q-network","title":"Amortized Variational Deep Q Network","date":"2020-11-03","arxiv_id":"2011.01706","repositories_listed":1,"syntology":null},{"url":"/paper/causal-campbell-goodhart-s-law-and","slug":"causal-campbell-goodhart-s-law-and","title":"Causal Campbell-Goodhart's law and Reinforcement Learning","date":"2020-11-02","arxiv_id":"2011.01010","repositories_listed":1,"syntology":null},{"url":"/paper/instance-based-generalization-in","slug":"instance-based-generalization-in","title":"Instance based Generalization in Reinforcement Learning","date":"2020-11-02","arxiv_id":"2011.01089","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":4,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":7,"phrase":"6 ran (of which 4 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/instance-based-generalization-in#ran","syntology_url":"https://syntology.ai/paper/2011.01089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.01089"}},"official":{"repos":["MartinBertran/InstanceAgnosticPolicyEnsembles"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":4,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/self-driving-network-and-service-coordination","slug":"self-driving-network-and-service-coordination","title":"Self-Driving Network and Service Coordination Using Deep Reinforcement Learning","date":"2020-11-02","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/implicit-under-parameterization-inhibits-data-1","slug":"implicit-under-parameterization-inhibits-data-1","title":"Implicit Under-Parameterization Inhibits Data-Efficient Deep Reinforcement Learning","date":"2020-10-27","arxiv_id":"2010.14498","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/implicit-under-parameterization-inhibits-data-1#ran","syntology_url":"https://syntology.ai/paper/2010.14498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.14498"}},"official":null}},{"url":"/paper/learning-financial-asset-specific-trading","slug":"learning-financial-asset-specific-trading","title":"Learning Financial Asset-Specific Trading Rules via Deep Reinforcement Learning","date":"2020-10-27","arxiv_id":"2010.14194","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-enhanced-heterogeneous","slug":"reinforcement-learning-enhanced-heterogeneous","title":"Personalised Meta-path Generation for Heterogeneous GNNs","date":"2020-10-26","arxiv_id":"2010.13735","repositories_listed":1,"syntology":null},{"url":"/paper/how-to-make-deep-rl-work-in-practice","slug":"how-to-make-deep-rl-work-in-practice","title":"How to Make Deep RL Work in Practice","date":"2020-10-25","arxiv_id":"2010.13083","repositories_listed":1,"syntology":null},{"url":"/paper/bridging-imagination-and-reality-for-model","slug":"bridging-imagination-and-reality-for-model","title":"Bridging Imagination and Reality for Model-Based Deep Reinforcement Learning","date":"2020-10-23","arxiv_id":"2010.12142","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/bridging-imagination-and-reality-for-model#ran","syntology_url":"https://syntology.ai/paper/2010.12142","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.12142"}},"official":{"repos":["Mehooz/BIRD_code"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/multi-uav-path-planning-for-wireless-data","slug":"multi-uav-path-planning-for-wireless-data","title":"Multi-UAV Path Planning for Wireless Data Harvesting with Deep Reinforcement Learning","date":"2020-10-23","arxiv_id":"2010.12461","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-stacked","slug":"deep-reinforcement-learning-with-stacked","title":"Deep Reinforcement Learning with Stacked Hierarchical Attention for Text-based Games","date":"2020-10-22","arxiv_id":"2010.11655","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-reinforcement-learning-with-stacked#ran","syntology_url":"https://syntology.ai/paper/2010.11655","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.11655"}},"official":{"repos":["YunqiuXu/SHA-KG"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-with-combinatorial","slug":"reinforcement-learning-with-combinatorial","title":"Reinforcement Learning with Combinatorial Actions: An Application to Vehicle Routing","date":"2020-10-22","arxiv_id":"2010.12001","repositories_listed":1,"syntology":null},{"url":"/paper/correlation-aware-cooperative-multigroup","slug":"correlation-aware-cooperative-multigroup","title":"Correlation-aware Cooperative Multigroup Broadcast 360° Video Delivery Network: A Hierarchical Deep Reinforcement Learning Approach","date":"2020-10-21","arxiv_id":"2010.11347","repositories_listed":1,"syntology":null},{"url":"/paper/iterative-amortized-policy-optimization-1","slug":"iterative-amortized-policy-optimization-1","title":"Iterative Amortized Policy Optimization","date":"2020-10-20","arxiv_id":"2010.10670","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/iterative-amortized-policy-optimization-1#ran","syntology_url":"https://syntology.ai/paper/2010.10670","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.10670"}},"official":{"repos":["joelouismarino/variational_rl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/deep-reinforcement-learning-with-population","slug":"deep-reinforcement-learning-with-population","title":"Deep Reinforcement Learning with Population-Coded Spiking Neural Network for Continuous Control","date":"2020-10-19","arxiv_id":"2010.09635","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/deep-reinforcement-learning-with-population#ran","syntology_url":"https://syntology.ai/paper/2010.09635","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.09635"}},"official":{"repos":["combra-lab/pop-spiking-deep-rl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/an-alternative-to-backpropagation-in-deep","slug":"an-alternative-to-backpropagation-in-deep","title":"MAP Propagation Algorithm: Faster Learning with a Team of Reinforcement Learning Agents","date":"2020-10-15","arxiv_id":"2010.07893","repositories_listed":1,"syntology":null},{"url":"/paper/multi-task-deep-reinforcement-learning-with-1","slug":"multi-task-deep-reinforcement-learning-with-1","title":"Knowledge Transfer in Multi-Task Deep Reinforcement Learning for Continuous Control","date":"2020-10-15","arxiv_id":"2010.07494","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-task-deep-reinforcement-learning-with-1#ran","syntology_url":"https://syntology.ai/paper/2010.07494","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.07494"}},"official":null}},{"url":"/paper/deep-reinforcement-learning-for-real-time","slug":"deep-reinforcement-learning-for-real-time","title":"Deep Reinforcement Learning for Real-Time Optimization of Pumps in Water Distribution Systems","date":"2020-10-13","arxiv_id":"2010.06460","repositories_listed":1,"syntology":null},{"url":"/paper/a-drl-based-multiagent-cooperative-control","slug":"a-drl-based-multiagent-cooperative-control","title":"A DRL-based Multiagent Cooperative Control Framework for CAV Networks: a Graphic Convolution Q Network","date":"2020-10-12","arxiv_id":"2010.05437","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-explanations-for-reinforcement-2","slug":"contrastive-explanations-for-reinforcement-2","title":"Contrastive Explanations for Reinforcement Learning via Embedded Self Predictions","date":"2020-10-11","arxiv_id":"2010.05180","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/contrastive-explanations-for-reinforcement-2#ran","syntology_url":"https://syntology.ai/paper/2010.05180","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2010.05180"}},"official":{"repos":["SuerpX/Embedded-Self-Predictions"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/distributed-resource-allocation-with-multi","slug":"distributed-resource-allocation-with-multi","title":"Distributed Resource Allocation with Multi-Agent Deep Reinforcement Learning for 5G-V2V Communication","date":"2020-10-11","arxiv_id":"2010.05290","repositories_listed":1,"syntology":null},{"url":"/paper/graph-convolutional-value-decomposition-in-1","slug":"graph-convolutional-value-decomposition-in-1","title":"Graph Convolutional Value Decomposition in Multi-Agent Reinforcement Learning","date":"2020-10-09","arxiv_id":"2010.04740","repositories_listed":1,"syntology":null},{"url":"/paper/information-driven-adaptive-sensing-based-on","slug":"information-driven-adaptive-sensing-based-on","title":"Information-Driven Adaptive Sensing Based on Deep Reinforcement Learning","date":"2020-10-08","arxiv_id":"2010.04112","repositories_listed":1,"syntology":null},{"url":"/paper/student-initiated-action-advising-via-advice","slug":"student-initiated-action-advising-via-advice","title":"Student-Initiated Action Advising via Advice Novelty","date":"2020-10-01","arxiv_id":"2010.00381","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-efficient","slug":"deep-reinforcement-learning-for-efficient","title":"Deep Reinforcement Learning for Efficient Measurement of Quantum Devices","date":"2020-09-30","arxiv_id":"2009.14825","repositories_listed":1,"syntology":null},{"url":"/paper/a-traffic-light-dynamic-control-algorithm","slug":"a-traffic-light-dynamic-control-algorithm","title":"A Traffic Light Dynamic Control Algorithm with Deep Reinforcement Learning Based on GNN Prediction","date":"2020-09-29","arxiv_id":"2009.14627","repositories_listed":1,"syntology":null},{"url":"/paper/pdlight-a-deep-reinforcement-learning-traffic","slug":"pdlight-a-deep-reinforcement-learning-traffic","title":"PDLight: A Deep Reinforcement Learning Traffic Light Control Algorithm with Pressure and Dynamic Light Duration","date":"2020-09-29","arxiv_id":"2009.13711","repositories_listed":1,"syntology":null},{"url":"/paper/an-automatic-cost-learning-framework-for","slug":"an-automatic-cost-learning-framework-for","title":"An Automatic Cost Learning Framework for Image Steganography Using Deep Reinforcement Learning","date":"2020-09-25","arxiv_id":null,"repositories_listed":1,"syntology":null}],"record_sha256":"b0a18c2d8ad2d7df169a6ddc8587a0fcf83570027e5879500fe54468f9fd89dc","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}