{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/5","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":59,"rows_per_page":100,"rows":[401,500],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/4","next":"/task/deep-reinforcement-learning/papers/6","papers":[{"url":"/paper/an-empirical-study-of-deep-reinforcement","slug":"an-empirical-study-of-deep-reinforcement","title":"An Empirical Study of Deep Reinforcement Learning in Continuing Tasks","date":"2025-01-12","arxiv_id":"2501.06937","repositories_listed":1,"syntology":null},{"url":"/paper/drl-based-medium-term-planning-of-renewable","slug":"drl-based-medium-term-planning-of-renewable","title":"DRL-Based Medium-Term Planning of Renewable-Integrated Self-Scheduling Cascaded Hydropower to Guide Wholesale Market Participation","date":"2025-01-08","arxiv_id":"2501.04839","repositories_listed":1,"syntology":null},{"url":"/paper/neural-dnf-mt-a-neuro-symbolic-approach-for","slug":"neural-dnf-mt-a-neuro-symbolic-approach-for","title":"Neural DNF-MT: A Neuro-symbolic Approach for Learning Interpretable and Editable Policies","date":"2025-01-07","arxiv_id":"2501.03888","repositories_listed":1,"syntology":null},{"url":"/paper/co-activation-graph-analysis-of-safety","slug":"co-activation-graph-analysis-of-safety","title":"Co-Activation Graph Analysis of Safety-Verified and Explainable Deep Reinforcement Learning Policies","date":"2025-01-06","arxiv_id":"2501.03142","repositories_listed":1,"syntology":null},{"url":"/paper/sim-to-real-transfer-for-mobile-robots-with","slug":"sim-to-real-transfer-for-mobile-robots-with","title":"Sim-to-Real Transfer for Mobile Robots with Reinforcement Learning: from NVIDIA Isaac Sim to Gazebo and Real ROS 2 Robots","date":"2025-01-06","arxiv_id":"2501.02902","repositories_listed":1,"syntology":null},{"url":"/paper/the-meta-representation-hypothesis","slug":"the-meta-representation-hypothesis","title":"Representation Convergence: Mutual Distillation is Secretly a Form of Regularization","date":"2025-01-05","arxiv_id":"2501.02481","repositories_listed":1,"syntology":null},{"url":"/paper/diversity-optimization-for-travelling","slug":"diversity-optimization-for-travelling","title":"Diversity Optimization for Travelling Salesman Problem via Deep Reinforcement Learning","date":"2025-01-01","arxiv_id":"2501.00884","repositories_listed":1,"syntology":null},{"url":"/paper/plug-and-play-ppo-an-adaptive-point-prompt","slug":"plug-and-play-ppo-an-adaptive-point-prompt","title":"Plug-and-Play PPO: An Adaptive Point Prompt Optimizer Making SAM Greater","date":"2025-01-01","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/efficient-and-scalable-deep-reinforcement","slug":"efficient-and-scalable-deep-reinforcement","title":"Efficient and Scalable Deep Reinforcement Learning for Mean Field Control Games","date":"2024-12-28","arxiv_id":"2501.00052","repositories_listed":1,"syntology":null},{"url":"/paper/numerical-solutions-of-fixed-points-in-two","slug":"numerical-solutions-of-fixed-points-in-two","title":"Numerical solutions of fixed points in two-dimensional Kuramoto-Sivashinsky equation expedited by reinforcement learning","date":"2024-12-27","arxiv_id":"2501.00046","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-representation-for-interactive","slug":"contrastive-representation-for-interactive","title":"Contrastive Representation for Interactive Recommendation","date":"2024-12-24","arxiv_id":"2412.18396","repositories_listed":1,"syntology":null},{"url":"/paper/dynamic-optimization-of-portfolio-allocation","slug":"dynamic-optimization-of-portfolio-allocation","title":"A Deep Reinforcement Learning Framework for Dynamic Portfolio Optimization: Evidence from China's Stock Market","date":"2024-12-24","arxiv_id":"2412.18563","repositories_listed":1,"syntology":null},{"url":"/paper/navigating-data-corruption-in-machine","slug":"navigating-data-corruption-in-machine","title":"Navigating Data Corruption in Machine Learning: Balancing Quality, Quantity, and Imputation Strategies","date":"2024-12-24","arxiv_id":"2412.18296","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-with-time-scale","slug":"deep-reinforcement-learning-with-time-scale","title":"Deep reinforcement learning with time-scale invariant memory","date":"2024-12-19","arxiv_id":"2412.15292","repositories_listed":1,"syntology":null},{"url":"/paper/spatio-temporal-sir-model-of-pandemic-spread","slug":"spatio-temporal-sir-model-of-pandemic-spread","title":"Spatio-Temporal SIR Model of Pandemic Spread During Warfare with Optimal Dual-use Healthcare System Administration using Deep Reinforcement Learning","date":"2024-12-18","arxiv_id":"2412.14039","repositories_listed":1,"syntology":null},{"url":"/paper/drl4aoi-a-drl-framework-for-semantic-aware","slug":"drl4aoi-a-drl-framework-for-semantic-aware","title":"DRL4AOI: A DRL Framework for Semantic-aware AOI Segmentation in Location-Based Services","date":"2024-12-06","arxiv_id":"2412.05437","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-an-overview","slug":"reinforcement-learning-an-overview","title":"Reinforcement Learning: An Overview","date":"2024-12-06","arxiv_id":"2412.05265","repositories_listed":1,"syntology":null},{"url":"/paper/gram-generalization-in-deep-rl-with-a-robust","slug":"gram-generalization-in-deep-rl-with-a-robust","title":"GRAM: Generalization in Deep RL with a Robust Adaptation Module","date":"2024-12-05","arxiv_id":"2412.04323","repositories_listed":1,"syntology":null},{"url":"/paper/pathletrl-optimizing-trajectory-pathlet","slug":"pathletrl-optimizing-trajectory-pathlet","title":"PathletRL++: Optimizing Trajectory Pathlet Extraction and Dictionary Formation via Reinforcement Learning","date":"2024-12-04","arxiv_id":"2412.03715","repositories_listed":1,"syntology":null},{"url":"/paper/conformal-symplectic-optimization-for-stable","slug":"conformal-symplectic-optimization-for-stable","title":"Conformal Symplectic Optimization for Stable Reinforcement Learning","date":"2024-12-03","arxiv_id":"2412.02291","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-plastic-waste-collection-in-water","slug":"optimizing-plastic-waste-collection-in-water","title":"Optimizing Plastic Waste Collection in Water Bodies Using Heterogeneous Autonomous Surface Vehicles with Deep Reinforcement Learning","date":"2024-12-03","arxiv_id":"2412.02316","repositories_listed":1,"syntology":null},{"url":"/paper/step-by-step-guidance-to-differential-anemia","slug":"step-by-step-guidance-to-differential-anemia","title":"Step-by-Step Guidance to Differential Anemia Diagnosis with Real-World Data and Deep Reinforcement Learning","date":"2024-12-03","arxiv_id":"2412.02273","repositories_listed":1,"syntology":null},{"url":"/paper/monocular-obstacle-avoidance-based-on-inverse","slug":"monocular-obstacle-avoidance-based-on-inverse","title":"Monocular Obstacle Avoidance Based on Inverse PPO for Fixed-wing UAVs","date":"2024-11-27","arxiv_id":"2411.18009","repositories_listed":1,"syntology":null},{"url":"/paper/continual-deep-reinforcement-learning-with","slug":"continual-deep-reinforcement-learning-with","title":"Continual Deep Reinforcement Learning with Task-Agnostic Policy Distillation","date":"2024-11-25","arxiv_id":"2411.16532","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/continual-deep-reinforcement-learning-with#ran","syntology_url":"https://syntology.ai/paper/2411.16532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16532"}},"official":{"repos":["wabbajack1/tapd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-policy-gradient-methods-without-batch","slug":"deep-policy-gradient-methods-without-batch","title":"Deep Policy Gradient Methods Without Batch Updates, Target Networks, or Replay Buffers","date":"2024-11-22","arxiv_id":"2411.15370","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-policy-gradient-methods-without-batch#ran","syntology_url":"https://syntology.ai/paper/2411.15370","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.15370"}},"official":{"repos":["gauthamvasan/avg"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/drl-based-optimization-for-aoi-and-energy","slug":"drl-based-optimization-for-aoi-and-energy","title":"DRL-Based Optimization for AoI and Energy Consumption in C-V2X Enabled IoV","date":"2024-11-20","arxiv_id":"2411.13104","repositories_listed":1,"syntology":null},{"url":"/paper/action-attentive-deep-reinforcement-learning","slug":"action-attentive-deep-reinforcement-learning","title":"Action-Attentive Deep Reinforcement Learning for Autonomous Alignment of Beamlines","date":"2024-11-19","arxiv_id":"2411.12183","repositories_listed":1,"syntology":null},{"url":"/paper/that-chip-has-sailed-a-critique-of-unfounded","slug":"that-chip-has-sailed-a-critique-of-unfounded","title":"That Chip Has Sailed: A Critique of Unfounded Skepticism Around AI for Chip Design","date":"2024-11-15","arxiv_id":"2411.10053","repositories_listed":1,"syntology":null},{"url":"/paper/tangled-program-graphs-as-an-alternative-to","slug":"tangled-program-graphs-as-an-alternative-to","title":"Tangled Program Graphs as an alternative to DRL-based control algorithms for UAVs","date":"2024-11-08","arxiv_id":"2411.05586","repositories_listed":1,"syntology":null},{"url":"/paper/learning-generalizable-policy-for-obstacle","slug":"learning-generalizable-policy-for-obstacle","title":"Learning Generalizable Policy for Obstacle-Aware Autonomous Drone Racing","date":"2024-11-06","arxiv_id":"2411.04246","repositories_listed":1,"syntology":null},{"url":"/paper/a-little-less-conversation-a-little-more","slug":"a-little-less-conversation-a-little-more","title":"A little less conversation, a little more action, please: Investigating the physical common-sense of LLMs in a 3D embodied environment","date":"2024-10-30","arxiv_id":"2410.23242","repositories_listed":1,"syntology":null},{"url":"/paper/human-readable-programs-as-actors-of","slug":"human-readable-programs-as-actors-of","title":"Human-Readable Programs as Actors of Reinforcement Learning Agents Using Critic-Moderated Evolution","date":"2024-10-29","arxiv_id":"2410.21940","repositories_listed":1,"syntology":null},{"url":"/paper/learning-successor-features-the-simple-way","slug":"learning-successor-features-the-simple-way","title":"Learning Successor Features the Simple Way","date":"2024-10-29","arxiv_id":"2410.22133","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-successor-features-the-simple-way#ran","syntology_url":"https://syntology.ai/paper/2410.22133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22133"}},"official":{"repos":["raymondchua/simple_successor_features"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-agents-for","slug":"deep-reinforcement-learning-agents-for","title":"Deep Reinforcement Learning Agents for Strategic Production Policies in Microeconomic Market Simulations","date":"2024-10-27","arxiv_id":"2410.20550","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-battery-storage-energy-arbitrage","slug":"enhancing-battery-storage-energy-arbitrage","title":"Enhancing Battery Storage Energy Arbitrage with Deep Reinforcement Learning and Time-Series Forecasting","date":"2024-10-25","arxiv_id":"2410.20005","repositories_listed":1,"syntology":null},{"url":"/paper/entity-based-reinforcement-learning-for","slug":"entity-based-reinforcement-learning-for","title":"Entity-based Reinforcement Learning for Autonomous Cyber Defence","date":"2024-10-23","arxiv_id":"2410.17647","repositories_listed":1,"syntology":null},{"url":"/paper/reinfier-and-reintrainer-verification-and","slug":"reinfier-and-reintrainer-verification-and","title":"Reinfier and Reintrainer: Verification and Interpretation-Driven Safe Deep Reinforcement Learning Frameworks","date":"2024-10-19","arxiv_id":"2410.15127","repositories_listed":1,"syntology":null},{"url":"/paper/benchmarking-deep-reinforcement-learning-for-1","slug":"benchmarking-deep-reinforcement-learning-for-1","title":"Benchmarking Deep Reinforcement Learning for Navigation in Denied Sensor Environments","date":"2024-10-18","arxiv_id":"2410.14616","repositories_listed":1,"syntology":null},{"url":"/paper/streaming-deep-reinforcement-learning-finally","slug":"streaming-deep-reinforcement-learning-finally","title":"Streaming Deep Reinforcement Learning Finally Works","date":"2024-10-18","arxiv_id":"2410.14606","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/streaming-deep-reinforcement-learning-finally#ran","syntology_url":"https://syntology.ai/paper/2410.14606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14606"}},"official":{"repos":["mohmdelsayed/streaming-drl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-reinforcement-learning-for-online-5","slug":"deep-reinforcement-learning-for-online-5","title":"Deep Reinforcement Learning for Online Optimal Execution Strategies","date":"2024-10-17","arxiv_id":"2410.13493","repositories_listed":1,"syntology":null},{"url":"/paper/improving-generalization-on-the-procgen","slug":"improving-generalization-on-the-procgen","title":"Improving Generalization on the ProcGen Benchmark with Simple Architectural Changes and Scale","date":"2024-10-13","arxiv_id":"2410.10905","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-generalization-on-the-procgen#ran","syntology_url":"https://syntology.ai/paper/2410.10905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10905"}},"official":{"repos":["anndvision/vsop-3d"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-natural-language-based-strategies","slug":"exploring-natural-language-based-strategies","title":"Exploring Natural Language-Based Strategies for Efficient Number Learning in Children through Reinforcement Learning","date":"2024-10-10","arxiv_id":"2410.08334","repositories_listed":1,"syntology":null},{"url":"/paper/generative-artificial-intelligence-gai-for","slug":"generative-artificial-intelligence-gai-for","title":"Generative Artificial Intelligence (GAI) for Mobile Communications: A Diffusion Model Perspective","date":"2024-10-08","arxiv_id":"2410.06389","repositories_listed":1,"syntology":null},{"url":"/paper/mitigating-adversarial-perturbations-for-deep","slug":"mitigating-adversarial-perturbations-for-deep","title":"Mitigating Adversarial Perturbations for Deep Reinforcement Learning via Vector Quantization","date":"2024-10-04","arxiv_id":"2410.03376","repositories_listed":1,"syntology":null},{"url":"/paper/lotus-learning-based-online-thermal-and","slug":"lotus-learning-based-online-thermal-and","title":"Lotus: learning-based online thermal and latency variation management for two-stage detectors on edge devices","date":"2024-10-01","arxiv_id":"2410.10847","repositories_listed":1,"syntology":null},{"url":"/paper/multi-robot-informative-path-planning-for","slug":"multi-robot-informative-path-planning-for","title":"Scalable Multi-Robot Informative Path Planning for Target Mapping via Deep Reinforcement Learning","date":"2024-09-25","arxiv_id":"2409.16967","repositories_listed":1,"syntology":null},{"url":"/paper/fedslate-a-federated-deep-reinforcement","slug":"fedslate-a-federated-deep-reinforcement","title":"FedSlate:A Federated Deep Reinforcement Learning Recommender System","date":"2024-09-23","arxiv_id":"2409.14872","repositories_listed":1,"syntology":null},{"url":"/paper/synthesizing-evolving-symbolic","slug":"synthesizing-evolving-symbolic","title":"Synthesizing Evolving Symbolic Representations for Autonomous Systems","date":"2024-09-18","arxiv_id":"2409.11756","repositories_listed":1,"syntology":null},{"url":"/paper/audio-driven-reinforcement-learning-for-head","slug":"audio-driven-reinforcement-learning-for-head","title":"Audio-Driven Reinforcement Learning for Head-Orientation in Naturalistic Environments","date":"2024-09-16","arxiv_id":"2409.10048","repositories_listed":1,"syntology":null},{"url":"/paper/learning-discrete-world-models-for-heuristic","slug":"learning-discrete-world-models-for-heuristic","title":"Learning Discrete World Models for Heuristic Search","date":"2024-09-14","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/design-optimization-of-nuclear-fusion-reactor","slug":"design-optimization-of-nuclear-fusion-reactor","title":"Design Optimization of Nuclear Fusion Reactor through Deep Reinforcement Learning","date":"2024-09-12","arxiv_id":"2409.08231","repositories_listed":1,"syntology":null},{"url":"/paper/double-successive-over-relaxation-q-learning","slug":"double-successive-over-relaxation-q-learning","title":"Double Successive Over-Relaxation Q-Learning with an Extension to Deep Reinforcement Learning","date":"2024-09-10","arxiv_id":"2409.06356","repositories_listed":1,"syntology":null},{"url":"/paper/one-policy-to-run-them-all-an-end-to-end","slug":"one-policy-to-run-them-all-an-end-to-end","title":"One Policy to Run Them All: an End-to-end Learning Approach to Multi-Embodiment Locomotion","date":"2024-09-10","arxiv_id":"2409.06366","repositories_listed":1,"syntology":null},{"url":"/paper/semifactual-explanations-for-reinforcement","slug":"semifactual-explanations-for-reinforcement","title":"Semifactual Explanations for Reinforcement Learning","date":"2024-09-09","arxiv_id":"2409.05435","repositories_listed":1,"syntology":null},{"url":"/paper/soft-actor-critic-with-beta-policy-via","slug":"soft-actor-critic-with-beta-policy-via","title":"Soft Actor-Critic with Beta Policy via Implicit Reparameterization Gradients","date":"2024-09-08","arxiv_id":"2409.04971","repositories_listed":1,"syntology":null},{"url":"/paper/improving-deep-reinforcement-learning-by","slug":"improving-deep-reinforcement-learning-by","title":"Improving Deep Reinforcement Learning by Reducing the Chain Effect of Value and Policy Churn","date":"2024-09-07","arxiv_id":"2409.04792","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":3,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 3 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-deep-reinforcement-learning-by#ran","syntology_url":"https://syntology.ai/paper/2409.04792","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.04792"}},"official":{"repos":["bluecontra/CHAIN"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":3,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-deep-reinforcement-learning-framework-for-8","slug":"a-deep-reinforcement-learning-framework-for-8","title":"A Deep Reinforcement Learning Framework For Financial Portfolio Management","date":"2024-09-03","arxiv_id":"2409.08426","repositories_listed":1,"syntology":null},{"url":"/paper/ai-olympics-challenge-with-evolutionary-soft","slug":"ai-olympics-challenge-with-evolutionary-soft","title":"AI Olympics challenge with Evolutionary Soft Actor Critic","date":"2024-09-02","arxiv_id":"2409.01104","repositories_listed":1,"syntology":null},{"url":"/paper/solving-integrated-process-planning-and","slug":"solving-integrated-process-planning-and","title":"Solving Integrated Process Planning and Scheduling Problem via Graph Neural Network Based Deep Reinforcement Learning","date":"2024-09-02","arxiv_id":"2409.00968","repositories_listed":1,"syntology":null},{"url":"/paper/aggym-an-agricultural-biotic-stress","slug":"aggym-an-agricultural-biotic-stress","title":"AgGym: An agricultural biotic stress simulation environment for ultra-precision management planning","date":"2024-09-01","arxiv_id":"2409.00735","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-camera-exposure-control-for-visual","slug":"efficient-camera-exposure-control-for-visual","title":"Efficient Camera Exposure Control for Visual Odometry via Deep Reinforcement Learning","date":"2024-08-30","arxiv_id":"2408.17005","repositories_listed":1,"syntology":null},{"url":"/paper/mapf-gpt-imitation-learning-for-multi-agent-1","slug":"mapf-gpt-imitation-learning-for-multi-agent-1","title":"MAPF-GPT: Imitation Learning for Multi-Agent Pathfinding at Scale","date":"2024-08-29","arxiv_id":"2409.00134","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/mapf-gpt-imitation-learning-for-multi-agent-1#ran","syntology_url":"https://syntology.ai/paper/2409.00134","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00134"}},"official":{"repos":["cognitiveaisystems/mapf-gpt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/control-informed-reinforcement-learning-for","slug":"control-informed-reinforcement-learning-for","title":"Control-Informed Reinforcement Learning for Chemical Processes","date":"2024-08-24","arxiv_id":"2408.13566","repositories_listed":1,"syntology":null},{"url":"/paper/dutytte-deciphering-uncertainty-in-origin","slug":"dutytte-deciphering-uncertainty-in-origin","title":"DutyTTE: Deciphering Uncertainty in Origin-Destination Travel Time Estimation","date":"2024-08-23","arxiv_id":"2408.12809","repositories_listed":1,"syntology":null},{"url":"/paper/hologram-reasoning-for-solving-algebra","slug":"hologram-reasoning-for-solving-algebra","title":"Hologram Reasoning for Solving Algebra Problems with Geometry Diagrams","date":"2024-08-20","arxiv_id":"2408.10592","repositories_listed":1,"syntology":null},{"url":"/paper/physics-aware-combinatorial-assembly-planning","slug":"physics-aware-combinatorial-assembly-planning","title":"Physics-Aware Combinatorial Assembly Sequence Planning using Data-free Action Masking","date":"2024-08-19","arxiv_id":"2408.10162","repositories_listed":1,"syntology":null},{"url":"/paper/drl-based-resource-allocation-for-motion-blur","slug":"drl-based-resource-allocation-for-motion-blur","title":"DRL-Based Resource Allocation for Motion Blur Resistant Federated Self-Supervised Learning in IoV","date":"2024-08-17","arxiv_id":"2408.09194","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-saliency-ranking-and","slug":"integrating-saliency-ranking-and","title":"Integrating Saliency Ranking and Reinforcement Learning for Enhanced Object Detection","date":"2024-08-13","arxiv_id":"2408.06803","repositories_listed":1,"syntology":null},{"url":"/paper/model-based-transfer-learning-for-contextual","slug":"model-based-transfer-learning-for-contextual","title":"Model-Based Transfer Learning for Contextual Reinforcement Learning","date":"2024-08-08","arxiv_id":"2408.04498","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/model-based-transfer-learning-for-contextual#ran","syntology_url":"https://syntology.ai/paper/2408.04498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04498"}},"official":{"repos":["jhoon-cho/mbtl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generalized-gaussian-temporal-difference","slug":"generalized-gaussian-temporal-difference","title":"Generalized Gaussian Temporal Difference Error for Uncertainty-aware Reinforcement Learning","date":"2024-08-05","arxiv_id":"2408.02295","repositories_listed":1,"syntology":null},{"url":"/paper/2407-21565","slug":"2407-21565","title":"Multi-agent reinforcement learning for the control of three-dimensional Rayleigh-Bénard convection","date":"2024-07-31","arxiv_id":"2407.21565","repositories_listed":1,"syntology":null},{"url":"/paper/black-box-meta-learning-intrinsic-rewards-for","slug":"black-box-meta-learning-intrinsic-rewards-for","title":"Black box meta-learning intrinsic rewards for sparse-reward environments","date":"2024-07-31","arxiv_id":"2407.21546","repositories_listed":1,"syntology":null},{"url":"/paper/navix-scaling-minigrid-environments-with-jax","slug":"navix-scaling-minigrid-environments-with-jax","title":"NAVIX: Scaling MiniGrid Environments with JAX","date":"2024-07-28","arxiv_id":"2407.19396","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/navix-scaling-minigrid-environments-with-jax#ran","syntology_url":"https://syntology.ai/paper/2407.19396","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.19396"}},"official":{"repos":["epignatelli/navix"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/advanced-deep-reinforcement-learning-methods","slug":"advanced-deep-reinforcement-learning-methods","title":"Advanced deep-reinforcement-learning methods for flow control: group-invariant and positional-encoding networks improve learning speed and quality","date":"2024-07-25","arxiv_id":"2407.17822","repositories_listed":1,"syntology":null},{"url":"/paper/a-comparative-study-of-deep-reinforcement-1","slug":"a-comparative-study-of-deep-reinforcement-1","title":"A Comparative Study of Deep Reinforcement Learning Models: DQN vs PPO vs A2C","date":"2024-07-19","arxiv_id":"2407.14151","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-post-hoc-portfolio-management","slug":"explainable-post-hoc-portfolio-management","title":"Explainable Post hoc Portfolio Management Financial Policy of a Deep Reinforcement Learning agent","date":"2024-07-19","arxiv_id":"2407.14486","repositories_listed":1,"syntology":null},{"url":"/paper/instance-selection-for-dynamic-algorithm","slug":"instance-selection-for-dynamic-algorithm","title":"Instance Selection for Dynamic Algorithm Configuration with Reinforcement Learning: Improving Generalization","date":"2024-07-18","arxiv_id":"2407.13513","repositories_listed":1,"syntology":null},{"url":"/paper/reconfigurable-intelligent-surface-aided-21","slug":"reconfigurable-intelligent-surface-aided-21","title":"Reconfigurable Intelligent Surface Aided Vehicular Edge Computing: Joint Phase-shift Optimization and Multi-User Power Allocation","date":"2024-07-18","arxiv_id":"2407.13123","repositories_listed":1,"syntology":null},{"url":"/paper/joint-optimization-of-age-of-information-and","slug":"joint-optimization-of-age-of-information-and","title":"Joint Optimization of Age of Information and Energy Consumption in NR-V2X System based on Deep Reinforcement Learning","date":"2024-07-11","arxiv_id":"2407.08458","repositories_listed":1,"syntology":null},{"url":"/paper/cm-dqn-a-value-based-deep-reinforcement","slug":"cm-dqn-a-value-based-deep-reinforcement","title":"CM-DQN: A Value-Based Deep Reinforcement Learning Model to Simulate Confirmation Bias","date":"2024-07-10","arxiv_id":"2407.07454","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-for-sequential","slug":"deep-reinforcement-learning-for-sequential","title":"Deep Reinforcement Learning for Sequential Combinatorial Auctions","date":"2024-07-10","arxiv_id":"2407.08022","repositories_listed":1,"syntology":null},{"url":"/paper/resource-allocation-for-twin-maintenance-and","slug":"resource-allocation-for-twin-maintenance-and","title":"Resource Allocation for Twin Maintenance and Computing Task Processing in Digital Twin Vehicular Edge Computing Network","date":"2024-07-10","arxiv_id":"2407.07575","repositories_listed":1,"syntology":null},{"url":"/paper/economic-span-selection-of-bridge-based-on","slug":"economic-span-selection-of-bridge-based-on","title":"Economic span selection of bridge based on deep reinforcement learning","date":"2024-07-09","arxiv_id":"2407.06507","repositories_listed":1,"syntology":null},{"url":"/paper/graph-neural-networks-and-deep-reinforcement","slug":"graph-neural-networks-and-deep-reinforcement","title":"Graph Neural Networks and Deep Reinforcement Learning Based Resource Allocation for V2X Communications","date":"2024-07-09","arxiv_id":"2407.06518","repositories_listed":1,"syntology":null},{"url":"/paper/safe-and-reliable-training-of-learning-based","slug":"safe-and-reliable-training-of-learning-based","title":"Safe and Reliable Training of Learning-Based Aerospace Controllers","date":"2024-07-09","arxiv_id":"2407.07088","repositories_listed":1,"syntology":null},{"url":"/paper/fedmrl-data-heterogeneity-aware-federated","slug":"fedmrl-data-heterogeneity-aware-federated","title":"FedMRL: Data Heterogeneity Aware Federated Multi-agent Deep Reinforcement Learning for Medical Imaging","date":"2024-07-08","arxiv_id":"2407.05800","repositories_listed":1,"syntology":null},{"url":"/paper/safety-driven-deep-reinforcement-learning","slug":"safety-driven-deep-reinforcement-learning","title":"Safety-Driven Deep Reinforcement Learning Framework for Cobots: A Sim2Real Approach","date":"2024-07-02","arxiv_id":"2407.02231","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-age-of-information-in-vehicular","slug":"optimizing-age-of-information-in-vehicular","title":"Optimizing Age of Information in Vehicular Edge Computing with Federated Graph Neural Network Multi-Agent Reinforcement Learning","date":"2024-07-01","arxiv_id":"2407.02342","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-world-models-with-context-aware","slug":"efficient-world-models-with-context-aware","title":"Efficient World Models with Context-Aware Tokenization","date":"2024-06-27","arxiv_id":"2406.19320","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/efficient-world-models-with-context-aware#ran","syntology_url":"https://syntology.ai/paper/2406.19320","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19320"}},"official":{"repos":["vmicheli/delta-iris"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/breaking-the-barrier-enhanced-utility-and","slug":"breaking-the-barrier-enhanced-utility-and","title":"Breaking the Barrier: Enhanced Utility and Robustness in Smoothed DRL Agents","date":"2024-06-26","arxiv_id":"2406.18062","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":5,"n_violates":1,"n_no_contract":0,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 5 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/breaking-the-barrier-enhanced-utility-and#ran","syntology_url":"https://syntology.ai/paper/2406.18062","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18062"}},"official":{"repos":["trustworthy-ml-lab/robust_highutil_smoothed_drl"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/combining-automated-optimisation-of","slug":"combining-automated-optimisation-of","title":"Combining Automated Optimisation of Hyperparameters and Reward Shape","date":"2024-06-26","arxiv_id":"2406.18293","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-consistency-of-hyper-parameter","slug":"on-the-consistency-of-hyper-parameter","title":"On the consistency of hyper-parameter selection in value-based deep reinforcement learning","date":"2024-06-25","arxiv_id":"2406.17523","repositories_listed":1,"syntology":null},{"url":"/paper/text-alpha-2-discovering-logical-formulaic","slug":"text-alpha-2-discovering-logical-formulaic","title":"$\\text{Alpha}^2$: Discovering Logical Formulaic Alphas using Deep Reinforcement Learning","date":"2024-06-24","arxiv_id":"2406.16505","repositories_listed":1,"syntology":null},{"url":"/paper/a-benchmark-study-of-deep-rl-methods-for","slug":"a-benchmark-study-of-deep-rl-methods-for","title":"A Benchmark Study of Deep-RL Methods for Maximum Coverage Problems over Graphs","date":"2024-06-20","arxiv_id":"2406.14697","repositories_listed":1,"syntology":null},{"url":"/paper/deep-reinforcement-learning-based-aoi-aware","slug":"deep-reinforcement-learning-based-aoi-aware","title":"Deep-Reinforcement-Learning-Based AoI-Aware Resource Allocation for RIS-Aided IoV Networks","date":"2024-06-17","arxiv_id":"2406.11245","repositories_listed":1,"syntology":null},{"url":"/paper/online-context-learning-for-socially","slug":"online-context-learning-for-socially","title":"Online Context Learning for Socially Compliant Navigation","date":"2024-06-17","arxiv_id":"2406.11495","repositories_listed":1,"syntology":null},{"url":"/paper/reconfigurable-intelligent-surface-assisted-28","slug":"reconfigurable-intelligent-surface-assisted-28","title":"Reconfigurable Intelligent Surface Assisted VEC Based on Multi-Agent Reinforcement Learning","date":"2024-06-17","arxiv_id":"2406.11318","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-reinforcement-learning-for-swarm","slug":"hierarchical-reinforcement-learning-for-swarm","title":"Hierarchical Reinforcement Learning for Swarm Confrontation with High Uncertainty","date":"2024-06-12","arxiv_id":"2406.07877","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-to-disentangle","slug":"reinforcement-learning-to-disentangle","title":"Reinforcement Learning to Disentangle Multiqubit Quantum States from Partial Observations","date":"2024-06-12","arxiv_id":"2406.07884","repositories_listed":1,"syntology":null},{"url":"/paper/failures-are-fated-but-can-be-faded","slug":"failures-are-fated-but-can-be-faded","title":"Failures Are Fated, But Can Be Faded: Characterizing and Mitigating Unwanted Behaviors in Large-Scale Vision and Language Models","date":"2024-06-11","arxiv_id":"2406.07145","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/failures-are-fated-but-can-be-faded#ran","syntology_url":"https://syntology.ai/paper/2406.07145","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07145"}},"official":{"repos":["somsagar07/FailureShiftRL"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}}],"record_sha256":"06319ab5af2331d34109b136aa41dcb006fbd15ca874e4863114e7564af1b1af","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}