{"url":"/method/td-gammon","slug":"td-gammon","name":"TD-Gammon","full_name":"TD-Gammon","full_name_withheld":false,"description_markdown":"**TD-Gammon** is a game-learning architecture for playing backgammon. It involves the use of a $TD\\left(\\lambda\\right)$ learning algorithm and a feedforward neural network.\r\n\r\nCredit: [Temporal Difference Learning and\r\nTD-Gammon](https://cling.csd.uwo.ca/cs346a/extra/tdgammon.pdf)","description_state":"present","introduced_year":1992,"introduced_by":{"title":null,"paper":null,"first_author":null,"n_authors":0,"url_abs":null,"archive_paper_url":null},"source":{"url":null,"title":null,"url_on_a_paper_host":false},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Reinforcement Learning","area_id":"reinforcement-learning","collection":"Board Game Models","url":"/methods/category/board-game-models","pwc_aliases":[]}],"n_papers_tagged":4,"archive_num_papers":4,"papers_newest_first":[{"paper":null,"title":"On-line Policy Improvement using Monte-Carlo Search","date":"2025-01-09","arxiv_id":"2501.05407","n_code_links":0,"syntology":null},{"paper":null,"title":"Model Predictive Control and Reinforcement Learning: A Unified Framework Based on Dynamic Programming","date":"2024-06-02","arxiv_id":"2406.00592","n_code_links":0,"syntology":null},{"paper":null,"title":"Search in Imperfect Information Games","date":"2021-11-10","arxiv_id":"2111.05884","n_code_links":0,"syntology":null},{"paper":null,"title":"A Hierarchical Reinforcement Learning Method for Persistent Time-Sensitive Tasks","date":"2016-06-20","arxiv_id":"1606.06355","n_code_links":0,"syntology":null}],"papers_shown":4,"tasks":[{"task":"/task/reinforcement-learning-1","name":"Reinforcement Learning (RL)","papers":2},{"task":"/task/hierarchical-reinforcement-learning","name":"Hierarchical Reinforcement Learning","papers":1},{"task":"/task/model-predictive-control","name":"Model Predictive Control","papers":1},{"task":"/task/reinforcement-learning","name":"Reinforcement Learning","papers":1},{"task":"/task/reinforcement-learning-2","name":"reinforcement-learning","papers":1}],"tasks_shown":5,"n_tasks":5,"usage_by_year":[{"year":"2016","papers":1},{"year":"2021","papers":1},{"year":"2024","papers":1},{"year":"2025","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/td-gammon"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}