{"url":"/method/gradientdice","slug":"gradientdice","name":"GradientDICE","full_name":"GradientDICE","full_name_withheld":false,"description_markdown":"**GradientDICE** is a density ratio learning method for estimating the density ratio between the state distribution of the target policy and the sampling distribution in off-policy reinforcement learning. It optimizes a different objective from [GenDICE](https://arxiv.org/abs/2002.09072) by using the Perron-Frobenius theorem and eliminating GenDICE’s use of divergence, such that nonlinearity in parameterization is not necessary for GradientDICE, which is provably convergent under linear function approximation.","description_state":"present","introduced_year":null,"introduced_by":{"title":"GradientDICE: Rethinking Generalized Offline Estimation of Stationary Values","paper":"/paper/gradientdice-rethinking-generalized-offline","first_author":"Shangtong Zhang","n_authors":3,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/gradientdice-rethinking-generalized-offline"},"source":{"url":"https://arxiv.org/abs/2001.11113v7","title":"GradientDICE: Rethinking Generalized Offline Estimation of Stationary Values","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Reinforcement Learning","area_id":"reinforcement-learning","collection":"Density Ratio Learning","url":"/methods/category/density-ratio-learning","pwc_aliases":[]}],"n_papers_tagged":1,"archive_num_papers":1,"papers_newest_first":[{"paper":"/paper/gradientdice-rethinking-generalized-offline","title":"GradientDICE: Rethinking Generalized Offline Estimation of Stationary Values","date":"2020-01-29","arxiv_id":"2001.11113","n_code_links":1,"syntology":null}],"papers_shown":1,"tasks":[{"task":"/task/reinforcement-learning","name":"Reinforcement Learning","papers":1}],"tasks_shown":1,"n_tasks":1,"usage_by_year":[{"year":"2020","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/gradientdice"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}