{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/deep-reinforcement-learning/papers/27","list_of":"/task/deep-reinforcement-learning","task":"Deep Reinforcement Learning","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":27,"pages_in_order":59,"rows_per_page":100,"rows":[2601,2700],"of":5822,"counts":{"archive_papers_tagged":5822,"with_a_code_link":1739,"where_syntology_ran_a_sample":398,"not_listed_spam_title":0,"listed":5822,"listed_where_code_ran":398,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":340,"every_run_a_failure_of_syntologys_instrument":58,"listed_with_a_run_with_no_instrument_failure":340,"listed_every_run_a_failure_of_syntologys_instrument":58,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/deep-reinforcement-learning","prev":"/task/deep-reinforcement-learning/papers/26","next":"/task/deep-reinforcement-learning/papers/28","papers":[{"url":null,"slug":"in-deep-reinforcement-learning-a-pruned","title":"In value-based deep reinforcement learning, a pruned network is a good network","date":"2024-02-19","arxiv_id":"2402.12479","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-mode-switching-and-resource-allocation","title":"Joint mode switching and resource allocation in wireless-powered RIS-aided multiuser communication systems","date":"2024-02-19","arxiv_id":"2402.12143","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimal-parallelization-strategies-for-active","title":"Optimal Parallelization Strategies for Active Flow Control in Deep Reinforcement Learning-Based Computational Fluid Dynamics","date":"2024-02-18","arxiv_id":"2402.11515","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-based-toolpath","title":"Deep Reinforcement Learning Based Toolpath Generation for Thermal Uniformity in Laser Powder Bed Fusion Process","date":"2024-02-17","arxiv_id":"2404.07209","repositories_listed":0,"syntology":null},{"url":null,"slug":"sinr-aware-deep-reinforcement-learning-for","title":"SINR-Aware Deep Reinforcement Learning for Distributed Dynamic Channel Allocation in Cognitive Interference Networks","date":"2024-02-17","arxiv_id":"2402.17773","repositories_listed":0,"syntology":null},{"url":null,"slug":"surpassing-legacy-approaches-and-human","title":"Surpassing legacy approaches to PWR core reload optimization with single-objective Reinforcement learning","date":"2024-02-16","arxiv_id":"2402.11040","repositories_listed":0,"syntology":null},{"url":null,"slug":"aligning-crowd-feedback-via-distributional","title":"Aligning Crowd Feedback via Distributional Preference Reward Modeling","date":"2024-02-15","arxiv_id":"2402.09764","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-courier-scheduling-in-crowdsourced","title":"Enhancing Courier Scheduling in Crowdsourced Last-Mile Delivery through Dynamic Shift Extensions: A Deep Reinforcement Learning Approach","date":"2024-02-15","arxiv_id":"2402.09961","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-experience-replayable-conditions","title":"Revisiting Experience Replayable Conditions","date":"2024-02-15","arxiv_id":"2402.10374","repositories_listed":0,"syntology":null},{"url":null,"slug":"drl-based-orchestration-of-multi-user-miso","title":"DRL-Based Orchestration of Multi-User MISO Systems with Stacked Intelligent Metasurfaces","date":"2024-02-14","arxiv_id":"2402.09006","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-estimation-bias-in-deep-double-q","title":"Exploiting Estimation Bias in Clipped Double Q-Learning for Continous Control Reinforcement Learning Tasks","date":"2024-02-14","arxiv_id":"2402.09078","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-enabled-flexible-job-shop-scheduling","title":"Learning-enabled Flexible Job-shop Scheduling for Scalable Smart Manufacturing","date":"2024-02-14","arxiv_id":"2402.08979","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-interpretable-policies-in-hindsight","title":"Learning Interpretable Policies in Hindsight-Observable POMDPs through Partially Supervised Reinforcement Learning","date":"2024-02-14","arxiv_id":"2402.09290","repositories_listed":0,"syntology":null},{"url":null,"slug":"ll-gabr-energy-efficient-live-video-streaming","title":"LL-GABR: Energy Efficient Live Video Streaming Using Reinforcement Learning","date":"2024-02-14","arxiv_id":"2402.09392","repositories_listed":0,"syntology":null},{"url":null,"slug":"single-reset-divide-conquer-imitation","title":"Single-Reset Divide & Conquer Imitation Learning","date":"2024-02-14","arxiv_id":"2402.09355","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-transient-stability","title":"Uncertainty-Aware Transient Stability-Constrained Preventive Redispatch: A Distributional Reinforcement Learning Approach","date":"2024-02-14","arxiv_id":"2402.09263","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-fair-and-firm-real-time-scheduling-in","title":"Towards Fair and Firm Real-Time Scheduling in DNN Multi-Tenant Multi-Accelerator Systems via Reinforcement Learning","date":"2024-02-09","arxiv_id":"2403.00766","repositories_listed":0,"syntology":null},{"url":null,"slug":"edge-caching-based-on-deep-reinforcement","title":"Attention-Enhanced Prioritized Proximal Policy Optimization for Adaptive Edge Caching","date":"2024-02-08","arxiv_id":"2402.14576","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-mode-switching-framework-for","title":"Intelligent Mode-switching Framework for Teleoperation","date":"2024-02-08","arxiv_id":"2402.06047","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-world-fluid-directed-rigid-body-control","title":"Real-World Fluid Directed Rigid Body Control via Deep Reinforcement Learning","date":"2024-02-08","arxiv_id":"2402.06102","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-artificial-intelligence-for-digital","title":"Scaling Artificial Intelligence for Digital Wargaming in Support of Decision-Making","date":"2024-02-08","arxiv_id":"2402.06075","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-intelligent-agents-in-combat","title":"Scaling Intelligent Agents in Combat Simulations for Wargaming","date":"2024-02-08","arxiv_id":"2402.06694","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-computational-approach-to-visual-ecology","title":"A computational approach to visual ecology with deep reinforcement learning","date":"2024-02-07","arxiv_id":"2402.05266","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-reinforcement-learning-approach-for-17","title":"A Deep Reinforcement Learning Approach for Adaptive Traffic Routing in Next-gen Networks","date":"2024-02-07","arxiv_id":"2402.04515","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-adversarial-inputs-in-deep","title":"Analyzing Adversarial Inputs in Deep Reinforcement Learning","date":"2024-02-07","arxiv_id":"2402.05284","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-computing-in-non-terrestrial","title":"Collaborative Computing in Non-Terrestrial Networks: A Multi-Time-Scale Deep Reinforcement Learning Approach","date":"2024-02-07","arxiv_id":"2402.04865","repositories_listed":0,"syntology":null},{"url":null,"slug":"compressing-deep-reinforcement-learning","title":"Compressing Deep Reinforcement Learning Networks with a Dynamic Structured Pruning Method for Autonomous Driving","date":"2024-02-07","arxiv_id":"2402.05146","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-deep-reinforcement-learning-for-2","title":"Collaborative Deep Reinforcement Learning for Resource Optimization in Non-Terrestrial Networks","date":"2024-02-06","arxiv_id":"2402.04056","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-picker","title":"Deep Reinforcement Learning for Picker Routing Problem in Warehousing","date":"2024-02-05","arxiv_id":"2402.03525","repositories_listed":0,"syntology":null},{"url":null,"slug":"frugal-actor-critic-sample-efficient-off","title":"Frugal Actor-Critic: Sample Efficient Off-Policy Deep Reinforcement Learning Using Unique Experiences","date":"2024-02-05","arxiv_id":"2402.05963","repositories_listed":0,"syntology":null},{"url":null,"slug":"iced-zero-shot-transfer-in-reinforcement","title":"DRED: Zero-Shot Transfer in Reinforcement Learning via Data-Regularised Environment Design","date":"2024-02-05","arxiv_id":"2402.03479","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-safe-reinforcement-learning-driven-weights","title":"A Safe Reinforcement Learning driven Weights-varying Model Predictive Control for Autonomous Vehicle Motion Control","date":"2024-02-04","arxiv_id":"2402.02624","repositories_listed":0,"syntology":null},{"url":null,"slug":"device-scheduling-and-assignment-in","title":"Device Scheduling and Assignment in Hierarchical Federated Learning for Internet of Things","date":"2024-02-04","arxiv_id":"2402.02506","repositories_listed":0,"syntology":null},{"url":null,"slug":"evading-deep-learning-based-malware-detectors","title":"Evading Deep Learning-Based Malware Detectors via Obfuscation: A Deep Reinforcement Learning Approach","date":"2024-02-04","arxiv_id":"2402.02600","repositories_listed":0,"syntology":null},{"url":null,"slug":"interference-aware-emergent-random-access","title":"Interference-Aware Emergent Random Access Protocol for Downlink LEO Satellite Networks","date":"2024-02-04","arxiv_id":"2402.02350","repositories_listed":0,"syntology":null},{"url":null,"slug":"obstacle-avoidance-deep-reinforcement","title":"Integrating DeepRL with Robust Low-Level Control in Robotic Manipulators for Non-Repetitive Reaching Tasks","date":"2024-02-04","arxiv_id":"2402.02551","repositories_listed":0,"syntology":null},{"url":null,"slug":"drl-based-dynamic-channel-access-and-sclar","title":"DRL-Based Dynamic Channel Access and SCLAR Maximization for Networks Under Jamming","date":"2024-02-02","arxiv_id":"2402.01574","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-the-market-sentiment-based-ensemble","title":"Learning the Market: Sentiment-Based Ensemble Trading Agents","date":"2024-02-02","arxiv_id":"2402.01441","repositories_listed":0,"syntology":null},{"url":null,"slug":"parametric-task-map-elites","title":"Parametric-Task MAP-Elites","date":"2024-02-02","arxiv_id":"2402.01275","repositories_listed":0,"syntology":null},{"url":null,"slug":"alpharank-an-artificial-intelligence-approach","title":"AlphaRank: An Artificial Intelligence Approach for Ranking and Selection Problems","date":"2024-02-01","arxiv_id":"2402.00907","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-policy-style-transfer","title":"Neural Policy Style Transfer","date":"2024-02-01","arxiv_id":"2402.00677","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-reinforcement-learning-based-controller-to","title":"A Reinforcement Learning Based Controller to Minimize Forces on the Crutches of a Lower-Limb Exoskeleton","date":"2024-01-31","arxiv_id":"2402.00135","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-graph-for-multi-robot-social","title":"Attention Graph for Multi-Robot Social Navigation with Deep Reinforcement Learning","date":"2024-01-31","arxiv_id":"2401.17914","repositories_listed":0,"syntology":null},{"url":null,"slug":"circuit-partitioning-for-multi-core-quantum","title":"Circuit Partitioning for Multi-Core Quantum Architectures with Deep Reinforcement Learning","date":"2024-01-31","arxiv_id":"2401.17976","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-versatile-dynamic","title":"Reinforcement Learning for Versatile, Dynamic, and Robust Bipedal Locomotion Control","date":"2024-01-30","arxiv_id":"2401.16889","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-deep-q-network-based-on-radial-basis","title":"A Deep Q-Network Based on Radial Basis Functions for Multi-Echelon Inventory Management","date":"2024-01-29","arxiv_id":"2401.15872","repositories_listed":0,"syntology":null},{"url":null,"slug":"attentive-convolutional-deep-reinforcement","title":"Attentive Convolutional Deep Reinforcement Learning for Optimizing Solar-Storage Systems in Real-Time Electricity Markets","date":"2024-01-29","arxiv_id":"2401.15853","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-voltage","title":"Deep Reinforcement Learning for Voltage Control and Renewable Accommodation Using Spatial-Temporal Graph Information","date":"2024-01-29","arxiv_id":"2401.15848","repositories_listed":0,"syntology":null},{"url":null,"slug":"autonomous-vehicle-patrolling-through-deep","title":"Autonomous Vehicle Patrolling Through Deep Reinforcement Learning: Learning to Communicate and Cooperate","date":"2024-01-28","arxiv_id":"2402.10222","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-for-15","title":"Tacit algorithmic collusion in deep reinforcement learning guided price competition: A study using EV charge pricing game","date":"2024-01-25","arxiv_id":"2401.15108","repositories_listed":0,"syntology":null},{"url":null,"slug":"traffic-learning-and-proactive-uav-trajectory","title":"Traffic Learning and Proactive UAV Trajectory Planning for Data Uplink in Markovian IoT Models","date":"2024-01-24","arxiv_id":"2401.13827","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-based-simulators-for-the","title":"Deep Learning Based Simulators for the Phosphorus Removal Process Control in Wastewater Treatment via Deep Reinforcement Learning Algorithms","date":"2024-01-23","arxiv_id":"2401.12822","repositories_listed":0,"syntology":null},{"url":null,"slug":"introducing-petrirl-an-innovative-framework","title":"Introducing PetriRL: An Innovative Framework for JSSP Resolution Integrating Petri nets and Event-based Reinforcement Learning","date":"2024-01-23","arxiv_id":"2402.00046","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-from-language-oriented","title":"Knowledge Distillation from Language-Oriented to Emergent Communication for Multi-Agent Remote Control","date":"2024-01-23","arxiv_id":"2401.12624","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-deep-reinforcement-learning-with-3","title":"Multi-agent deep reinforcement learning with centralized training and decentralized execution for transportation infrastructure management","date":"2024-01-23","arxiv_id":"2401.12455","repositories_listed":0,"syntology":null},{"url":null,"slug":"viewport-prediction-bitrate-selection-and","title":"Viewport Prediction, Bitrate Selection, and Beamforming Design for THz-Enabled 360° Video Streaming","date":"2024-01-23","arxiv_id":"2401.13114","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-and-scalable-network-slicing-by","title":"Fast and Scalable Network Slicing by Integrating Deep Learning with Lagrangian Methods","date":"2024-01-22","arxiv_id":"2401.11731","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-agent-dynamic-relational-reasoning-for","title":"Multi-Agent Dynamic Relational Reasoning for Social Robot Navigation","date":"2024-01-22","arxiv_id":"2401.12275","repositories_listed":0,"syntology":null},{"url":null,"slug":"safe-and-generalized-end-to-end-autonomous","title":"Efficient and Generalized end-to-end Autonomous Driving System with Latent Deep Reinforcement Learning and Demonstrations","date":"2024-01-22","arxiv_id":"2401.11792","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-reinforcement-learning-for-5","title":"Constrained Reinforcement Learning for Adaptive Controller Synchronization in Distributed SDN","date":"2024-01-21","arxiv_id":"2403.08775","repositories_listed":0,"syntology":null},{"url":null,"slug":"madrl-based-uavs-trajectory-design-with-anti","title":"MADRL-based UAVs Trajectory Design with Anti-Collision Mechanism in Vehicular Networks","date":"2024-01-21","arxiv_id":"2402.03342","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-empowered","title":"Deep Reinforcement Learning Empowered Activity-Aware Dynamic Health Monitoring Systems","date":"2024-01-19","arxiv_id":"2401.10794","repositories_listed":0,"syntology":null},{"url":null,"slug":"episodic-reinforcement-learning-with-expanded","title":"Episodic Reinforcement Learning with Expanded State-reward Space","date":"2024-01-19","arxiv_id":"2401.10516","repositories_listed":0,"syntology":null},{"url":null,"slug":"tiny-multi-agent-drl-for-twins-migration-in","title":"Tiny Multi-Agent DRL for Twins Migration in UAV Metaverses: A Multi-Leader Multi-Follower Stackelberg Game Approach","date":"2024-01-18","arxiv_id":"2401.09680","repositories_listed":0,"syntology":null},{"url":null,"slug":"traffic-smoothing-controllers-for-autonomous","title":"Traffic Smoothing Controllers for Autonomous Vehicles Using Deep Reinforcement Learning and Real-World Trajectory Data","date":"2024-01-18","arxiv_id":"2401.09666","repositories_listed":0,"syntology":null},{"url":null,"slug":"cnn-drl-with-shuffled-features-in-finance","title":"CNN-DRL with Shuffled Features in Finance","date":"2024-01-16","arxiv_id":"2402.03338","repositories_listed":0,"syntology":null},{"url":null,"slug":"cyclight-learning-traffic-signal-cooperation","title":"CycLight: learning traffic signal cooperation with a cycle-level strategy","date":"2024-01-16","arxiv_id":"2401.08121","repositories_listed":0,"syntology":null},{"url":null,"slug":"sum-throughput-maximization-in-multi-bd","title":"Sum Throughput Maximization in Multi-BD Symbiotic Radio NOMA Network Assisted by Active-STAR-RIS","date":"2024-01-16","arxiv_id":"2401.08301","repositories_listed":0,"syntology":null},{"url":null,"slug":"constrained-multi-objective-optimization-with","title":"Constrained Multi-objective Optimization with Deep Reinforcement Learning Assisted Operator Selection","date":"2024-01-15","arxiv_id":"2402.12381","repositories_listed":0,"syntology":null},{"url":null,"slug":"learned-best-effort-llm-serving","title":"Learned Best-Effort LLM Serving","date":"2024-01-15","arxiv_id":"2401.07886","repositories_listed":0,"syntology":null},{"url":null,"slug":"bet-explaining-deep-reinforcement-learning","title":"BET: Explaining Deep Reinforcement Learning through The Error-Prone Decisions","date":"2024-01-14","arxiv_id":"2401.07263","repositories_listed":0,"syntology":null},{"url":null,"slug":"ai-enabled-priority-and-auction-based","title":"AI-enabled Priority and Auction-Based Spectrum Management for 6G","date":"2024-01-12","arxiv_id":"2401.06484","repositories_listed":0,"syntology":null},{"url":null,"slug":"open-ran-lstm-traffic-prediction-and-slice","title":"Open RAN LSTM Traffic Prediction and Slice Management using Deep Reinforcement Learning","date":"2024-01-12","arxiv_id":"2401.06922","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatial-aware-deep-reinforcement-learning-for","title":"Spatial-Aware Deep Reinforcement Learning for the Traveling Officer Problem","date":"2024-01-11","arxiv_id":"2401.05969","repositories_listed":0,"syntology":null},{"url":null,"slug":"modelling-positioning-and-deep-reinforcement","title":"Modelling, Positioning, and Deep Reinforcement Learning Path Tracking Control of Scaled Robotic Vehicles: Design and Experimental Validation","date":"2024-01-10","arxiv_id":"2401.05194","repositories_listed":0,"syntology":null},{"url":null,"slug":"react-reinforcement-learning-for-controller","title":"ReACT: Reinforcement Learning for Controller Parametrization using B-Spline Geometries","date":"2024-01-10","arxiv_id":"2401.05251","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-safe-load-balancing-based-on-control","title":"Towards Safe Load Balancing based on Control Barrier Functions and Deep Reinforcement Learning","date":"2024-01-10","arxiv_id":"2401.05525","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-multi-agent-learning","title":"Deep Reinforcement Multi-agent Learning framework for Information Gathering with Local Gaussian Processes for Water Monitoring","date":"2024-01-09","arxiv_id":"2401.04631","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-spiking-actor-network-with-intra-layer","title":"Fully Spiking Actor Network with Intra-layer Connections for Reinforcement Learning","date":"2024-01-09","arxiv_id":"2401.05444","repositories_listed":0,"syntology":null},{"url":"/paper/i-rebalance-personalized-vehicle","slug":"i-rebalance-personalized-vehicle","title":"i-Rebalance: Personalized Vehicle Repositioning for Supply Demand Balance","date":"2024-01-09","arxiv_id":"2401.04429","repositories_listed":0,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/i-rebalance-personalized-vehicle#ran","syntology_url":"https://syntology.ai/paper/2401.04429","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04429"}},"official":null}},{"url":null,"slug":"deep-reinforcement-learning-for-multi-truck","title":"Deep Reinforcement Learning for Multi-Truck Vehicle Routing Problems with Multi-Leg Demand Routes","date":"2024-01-08","arxiv_id":"2401.08669","repositories_listed":0,"syntology":null},{"url":null,"slug":"guiding-drones-by-information-gain","title":"Guiding drones by information gain","date":"2024-01-08","arxiv_id":"2401.03947","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-once-plan-arbitrarily-lopa-attention","title":"Learn Once Plan Arbitrarily (LOPA): Attention-Enhanced Deep Reinforcement Learning Method for Global Path Planning","date":"2024-01-08","arxiv_id":"2401.04145","repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-learning-via-dqn-for-log","title":"Semi-supervised learning via DQN for log anomaly detection","date":"2024-01-06","arxiv_id":"2401.03151","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-local-path","title":"Deep Reinforcement Learning for Local Path Following of an Autonomous Formula SAE Vehicle","date":"2024-01-05","arxiv_id":"2401.02903","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-analyzing-generalization-in-deep","title":"A Survey Analyzing Generalization in Deep Reinforcement Learning","date":"2024-01-04","arxiv_id":"2401.02349","repositories_listed":0,"syntology":null},{"url":null,"slug":"ofdm-based-digital-semantic-communication","title":"OFDM-Based Digital Semantic Communication with Importance Awareness","date":"2024-01-04","arxiv_id":"2401.02178","repositories_listed":0,"syntology":null},{"url":null,"slug":"trajectory-oriented-policy-optimization-with","title":"Trajectory-Oriented Policy Optimization with Sparse Rewards","date":"2024-01-04","arxiv_id":"2401.02225","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-based-agricultural-management-in","title":"Learning-based agricultural management in partially observable environments subject to climate variability","date":"2024-01-02","arxiv_id":"2401.01273","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforcement-learning-for-sar-view-angle","title":"Reinforcement Learning for SAR View Angle Inversion with Differentiable SAR Renderer","date":"2024-01-02","arxiv_id":"2401.01165","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-learning-based-agent-modeling-for","title":"Contrastive learning-based agent modeling for deep reinforcement learning","date":"2023-12-30","arxiv_id":"2401.00132","repositories_listed":0,"syntology":null},{"url":null,"slug":"policy-optimization-with-smooth-guidance","title":"Policy Optimization with Smooth Guidance Learned from State-Only Demonstrations","date":"2023-12-30","arxiv_id":"2401.00162","repositories_listed":0,"syntology":null},{"url":null,"slug":"hibid-a-cross-channel-constrained-bidding","title":"HiBid: A Cross-Channel Constrained Bidding System with Budget Allocation by Hierarchical Offline Deep Reinforcement Learning","date":"2023-12-29","arxiv_id":"2312.17503","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-logo-deep-reinforcement-learning","title":"RL-LOGO: Deep Reinforcement Learning Localization for Logo Recognition","date":"2023-12-28","arxiv_id":"2312.16792","repositories_listed":0,"syntology":null},{"url":null,"slug":"risk-anticipatory-autonomous-driving","title":"Risk-anticipatory autonomous driving strategies considering vehicles' weights, based on hierarchical deep reinforcement learning","date":"2023-12-27","arxiv_id":"2401.08661","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-spatial-attention-and-proprioceptive","title":"Visual Spatial Attention and Proprioceptive Data-Driven Reinforcement Learning for Robust Peg-in-Hole Task Under Variable Conditions","date":"2023-12-27","arxiv_id":"2312.16438","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bayesian-framework-of-deep-reinforcement","title":"A Bayesian Framework of Deep Reinforcement Learning for Joint O-RAN/MEC Orchestration","date":"2023-12-26","arxiv_id":"2312.16142","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-kalman-based-hybrid-car-following","title":"Adaptive Kalman-based hybrid car following strategy using TD3 and CACC","date":"2023-12-26","arxiv_id":"2312.15993","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-target-detection-algorithm-in-traffic","title":"A Target Detection Algorithm in Traffic Scenes Based on Deep Reinforcement Learning","date":"2023-12-25","arxiv_id":"2312.15606","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-reinforcement-learning-for-quantitative","title":"Deep Reinforcement Learning for Quantitative Trading","date":"2023-12-25","arxiv_id":"2312.15730","repositories_listed":0,"syntology":null}],"record_sha256":"621aac7833d0e0f2f4fcb2ad1795ea9b7111ad1ca70c57632abdedbf64260b3b","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}