{"url":"/method/stochastic-dueling-network","slug":"stochastic-dueling-network","name":"Stochastic Dueling Network","full_name":"Stochastic Dueling Network","full_name_withheld":false,"description_markdown":"A **Stochastic Dueling Network**, or **SDN**, is an architecture for learning a value function $V$. The SDN learns both $V$ and $Q$ off-policy while maintaining consistency between the two estimates. At each time step it outputs a stochastic estimate of $Q$ and a deterministic estimate of $V$.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Sample Efficient Actor-Critic with Experience Replay","paper":"/paper/sample-efficient-actor-critic-with-experience","first_author":"Ziyu Wang","n_authors":7,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/sample-efficient-actor-critic-with-experience"},"source":{"url":"http://arxiv.org/abs/1611.01224v2","title":"Sample Efficient Actor-Critic with Experience Replay","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Reinforcement Learning","area_id":"reinforcement-learning","collection":"Value Function Estimation","url":"/methods/category/value-function-estimation","pwc_aliases":[]}],"n_papers_tagged":12,"archive_num_papers":12,"papers_newest_first":[{"paper":null,"title":"Dynamics of Resource Allocation in O-RANs: An In-depth Exploration of On-Policy and Off-Policy Deep Reinforcement Learning for Real-Time Applications","date":"2024-11-17","arxiv_id":"2412.01839","n_code_links":0,"syntology":null},{"paper":"/paper/joint-physical-digital-facial-attack","title":"Joint Physical-Digital Facial Attack Detection Via Simulating Spoofing Clues","date":"2024-04-12","arxiv_id":"2404.08450","n_code_links":3,"syntology":null},{"paper":null,"title":"Distributional Estimation of Data Uncertainty for Surveillance Face Anti-spoofing","date":"2023-09-18","arxiv_id":"2309.09485","n_code_links":0,"syntology":null},{"paper":null,"title":"PDVN: A Patch-based Dual-view Network for Face Liveness Detection using Light Field Focal Stack","date":"2023-01-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"title":"Asynchronous Curriculum Experience Replay: A Deep Reinforcement Learning Approach for UAV Autonomous Motion Control in Unknown Dynamic Environments","date":"2022-07-04","arxiv_id":"2207.01251","n_code_links":0,"syntology":null},{"paper":null,"title":"Is Word Error Rate a good evaluation metric for Speech Recognition in Indic Languages?","date":"2022-03-30","arxiv_id":"2203.16601","n_code_links":0,"syntology":null},{"paper":null,"title":"Learning Reward Machines: A Study in Partially Observable Reinforcement Learning","date":"2021-12-17","arxiv_id":"2112.09477","n_code_links":0,"syntology":null},{"paper":"/paper/a-deeppixbis-attentional-angular-margin-for","title":"A-DeepPixBis: Attentional Angular Margin for Face Anti-Spoofing","date":"2021-03-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/learning-reward-machines-for-partially","title":"Learning Reward Machines for Partially Observable Reinforcement Learning","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"title":"Sample Efficient Deep Reinforcement Learning for Dialogue Systems with Large Action Spaces","date":"2018-02-11","arxiv_id":"1802.03753","n_code_links":0,"syntology":null},{"paper":null,"title":"Pretraining Deep Actor-Critic Reinforcement Learning Algorithms With Expert Demonstrations","date":"2018-01-31","arxiv_id":"1801.10459","n_code_links":0,"syntology":null},{"paper":"/paper/sample-efficient-actor-critic-with-experience","title":"Sample Efficient Actor-Critic with Experience Replay","date":"2016-11-03","arxiv_id":"1611.01224","n_code_links":7,"syntology":{"ran":0,"of":1,"unverified":1,"pointer_only":1}}],"papers_shown":12,"tasks":[{"task":"/task/deep-reinforcement-learning","name":"Deep Reinforcement Learning","papers":5},{"task":"/task/reinforcement-learning","name":"Reinforcement Learning","papers":5},{"task":"/task/reinforcement-learning-1","name":"Reinforcement Learning (RL)","papers":5},{"task":"/task/reinforcement-learning-2","name":"reinforcement-learning","papers":5},{"task":"/task/face-anti-spoofing","name":"Face Anti-Spoofing","papers":3},{"task":"/task/face-recognition","name":"Face Recognition","papers":3},{"task":"/task/partially-observable-reinforcement-learning","name":"Partially Observable Reinforcement Learning","papers":2},{"task":"/task/problem-decomposition","name":"Problem Decomposition","papers":2},{"task":"/task/automatic-speech-recognition-2","name":"Automatic Speech Recognition","papers":1},{"task":"/task/automatic-speech-recognition","name":"Automatic Speech Recognition (ASR)","papers":1},{"task":"/task/benchmarking","name":"Benchmarking","papers":1},{"task":"/task/binary-classification","name":"Binary Classification","papers":1},{"task":"/task/continuous-control","name":"Continuous Control","papers":1},{"task":"/task/data-augmentation","name":"Data Augmentation","papers":1},{"task":"/task/face-presentation-attack-detection","name":"Face Presentation Attack Detection","papers":1},{"task":"/task/architecture-search","name":"Neural Architecture Search","papers":1},{"task":"/task/speech-recognition","name":"Speech Recognition","papers":1},{"task":"/task/spoken-dialogue-systems","name":"Spoken Dialogue Systems","papers":1},{"task":"/task/continuous-control","name":"continuous-control","papers":1},{"task":"/task/speech-recognition-1","name":"speech-recognition","papers":1}],"tasks_shown":20,"n_tasks":20,"usage_by_year":[{"year":"2016","papers":1},{"year":"2018","papers":2},{"year":"2019","papers":1},{"year":"2021","papers":2},{"year":"2022","papers":2},{"year":"2023","papers":2},{"year":"2024","papers":2}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/stochastic-dueling-network"},"syntology_read_at":"2026-09-25T09:33:49+00:00"}