{"url":"/method/epsilon-greedy-exploration","slug":"epsilon-greedy-exploration","name":"Epsilon Greedy Exploration","full_name":"Epsilon Greedy Exploration","full_name_withheld":false,"description_markdown":"**$\\epsilon$-Greedy Exploration** is an exploration strategy in reinforcement learning that takes an exploratory action with probability $\\epsilon$ and a greedy action with probability $1-\\epsilon$. It tackles the exploration-exploitation tradeoff with reinforcement learning algorithms: the desire to explore the state space with the desire to seek an optimal policy. Despite its simplicity, it is still commonly used as an behaviour policy $\\pi$ in several state-of-the-art reinforcement learning models.\r\n\r\nImage Credit: [Robin van Embden](https://cran.r-project.org/web/packages/contextual/vignettes/sutton_barto.html)","description_state":"present","introduced_year":null,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":null,"title":null,"url_on_a_paper_host":false},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Reinforcement Learning","area_id":"reinforcement-learning","collection":"Behaviour Policies","url":"/methods/category/behaviour-policies","pwc_aliases":[]}],"n_papers_tagged":5,"archive_num_papers":null,"papers_newest_first":[{"paper":null,"title":"Reinforcement Learning for an Efficient and Effective Malware Investigation during Cyber Incident Response","date":"2024-08-04","arxiv_id":"2408.01999","n_code_links":0,"syntology":null},{"paper":"/paper/sampling-efficient-deep-reinforcement","title":"Sampling Efficient Deep Reinforcement Learning through Preference-Guided Stochastic Exploration","date":"2022-06-20","arxiv_id":"2206.09627","n_code_links":1,"syntology":null},{"paper":null,"title":"Exploiting Semantic Epsilon Greedy Exploration Strategy in Multi-Agent Reinforcement Learning","date":"2022-01-26","arxiv_id":"2201.10803","n_code_links":0,"syntology":null},{"paper":null,"title":"Individual specialization in multi-task environments with multiagent reinforcement learners","date":"2019-12-29","arxiv_id":"1912.12671","n_code_links":0,"syntology":null},{"paper":"/paper/playing-atari-with-deep-reinforcement","title":"Playing Atari with Deep Reinforcement Learning","date":"2013-12-19","arxiv_id":"1312.5602","n_code_links":112,"syntology":{"ran":56,"of":117,"unverified":61,"pointer_only":56}}],"papers_shown":5,"tasks":[{"task":"/task/reinforcement-learning-1","name":"Reinforcement Learning (RL)","papers":5},{"task":"/task/q-learning","name":"Q-Learning","papers":3},{"task":"/task/reinforcement-learning","name":"Reinforcement Learning","papers":3},{"task":"/task/reinforcement-learning-2","name":"reinforcement-learning","papers":3},{"task":"/task/atari-games","name":"Atari Games","papers":2},{"task":"/task/deep-reinforcement-learning","name":"Deep Reinforcement Learning","papers":2},{"task":"/task/multi-agent-reinforcement-learning","name":"Multi-agent Reinforcement Learning","papers":2},{"task":"/task/decision-making","name":"Decision Making","papers":1},{"task":"/task/fairness","name":"Fairness","papers":1},{"task":"/task/malware-analysis","name":"Malware Analysis","papers":1},{"task":"/task/multi-goal-reinforcement-learning","name":"Multi-Goal Reinforcement Learning","papers":1},{"task":"/task/smac","name":"SMAC","papers":1},{"task":"/task/smac-1","name":"SMAC+","papers":1},{"task":"/task/starcraft","name":"Starcraft","papers":1}],"tasks_shown":14,"n_tasks":14,"usage_by_year":[{"year":"2013","papers":1},{"year":"2019","papers":1},{"year":"2022","papers":2},{"year":"2024","papers":1}],"row_source":"embedded","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/epsilon-greedy-exploration"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}