{"url":"/method/pwil","slug":"pwil","name":"PWIL","full_name":"Primal Wasserstein Imitation Learning","full_name_withheld":false,"description_markdown":"**Primal Wasserstein Imitation Learning**, or **PWIL**, is a method for imitation learning which ties to the primal form of the Wasserstein distance between the expert and the agent state-action distributions. The reward function is derived offline, as opposed to recent adversarial IL algorithms that learn a reward function through interactions with the environment, and requires little fine-tuning.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Primal Wasserstein Imitation Learning","paper":"/paper/primal-wasserstein-imitation-learning","first_author":"Robert Dadashi","n_authors":4,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/primal-wasserstein-imitation-learning"},"source":{"url":"https://arxiv.org/abs/2006.04678v2","title":"Primal Wasserstein Imitation Learning","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"Reinforcement Learning","area_id":"reinforcement-learning","collection":"Imitation Learning Methods","url":"/methods/category/imitation-learning-methods","pwc_aliases":[]}],"n_papers_tagged":3,"archive_num_papers":3,"papers_newest_first":[{"paper":null,"title":"Auto-Encoding Adversarial Imitation Learning","date":"2022-06-22","arxiv_id":"2206.11004","n_code_links":0,"syntology":null},{"paper":null,"title":"Auto-Encoding Inverse Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/primal-wasserstein-imitation-learning","title":"Primal Wasserstein Imitation Learning","date":"2020-06-08","arxiv_id":"2006.04678","n_code_links":2,"syntology":{"ran":0,"of":1,"unverified":1,"pointer_only":0}}],"papers_shown":3,"tasks":[{"task":"/task/imitation-learning","name":"Imitation Learning","papers":3},{"task":"/task/decision-making","name":"Decision Making","papers":2},{"task":"/task/mujoco","name":"MuJoCo","papers":2},{"task":"/task/reinforcement-learning-1","name":"Reinforcement Learning (RL)","papers":2},{"task":"/task/continuous-control","name":"Continuous Control","papers":1},{"task":"/task/reinforcement-learning","name":"Reinforcement Learning","papers":1},{"task":"/task/continuous-control","name":"continuous-control","papers":1},{"task":"/task/reinforcement-learning-2","name":"reinforcement-learning","papers":1}],"tasks_shown":8,"n_tasks":8,"usage_by_year":[{"year":"2020","papers":1},{"year":"2021","papers":1},{"year":"2022","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/pwil"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}