{"url":"/method/movement-pruning","slug":"movement-pruning","name":"Movement Pruning","full_name":"Movement Pruning","full_name_withheld":false,"description_markdown":"**Movement Pruning** is a simple, deterministic first-order weight pruning method that is more adaptive to pretrained model fine-tuning. Magnitude pruning can be seen as utilizing zeroth-order information (absolute value) of the running model. In contrast, movement pruning methods are where importance is derived from first-order information. Intuitively, instead of selecting weights that are far from zero, we retain connections that are moving away from zero during the training process.","description_state":"present","introduced_year":null,"introduced_by":{"title":"Movement Pruning: Adaptive Sparsity by Fine-Tuning","paper":"/paper/movement-pruning-adaptive-sparsity-by-fine","first_author":"Victor Sanh","n_authors":3,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/movement-pruning-adaptive-sparsity-by-fine"},"source":{"url":"https://arxiv.org/abs/2005.07683v2","title":"Movement Pruning: Adaptive Sparsity by Fine-Tuning","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Pruning","url":"/methods/category/pruning","pwc_aliases":[]}],"n_papers_tagged":5,"archive_num_papers":5,"papers_newest_first":[{"paper":null,"title":"On Importance of Pruning and Distillation for Efficient Low Resource NLP","date":"2024-09-21","arxiv_id":"2409.14162","n_code_links":0,"syntology":null},{"paper":"/paper/pruning-meets-low-rank-parameter-efficient","title":"LoRAPrune: Structured Pruning Meets Low-Rank Parameter-Efficient Fine-Tuning","date":"2023-05-28","arxiv_id":"2305.18403","n_code_links":1,"syntology":{"ran":0,"of":5,"unverified":5,"pointer_only":0}},{"paper":"/paper/what-matters-in-the-structured-pruning-of","title":"What Matters In The Structured Pruning of Generative Language Models?","date":"2023-02-07","arxiv_id":"2302.03773","n_code_links":1,"syntology":{"ran":0,"of":8,"unverified":8,"pointer_only":0}},{"paper":"/paper/block-pruning-for-faster-transformers","title":"Block Pruning For Faster Transformers","date":"2021-09-10","arxiv_id":"2109.04838","n_code_links":1,"syntology":null},{"paper":"/paper/movement-pruning-adaptive-sparsity-by-fine","title":"Movement Pruning: Adaptive Sparsity by Fine-Tuning","date":"2020-05-15","arxiv_id":"2005.07683","n_code_links":4,"syntology":{"ran":4,"of":4,"unverified":0,"pointer_only":4}}],"papers_shown":5,"tasks":[{"task":"/task/network-pruning","name":"Network Pruning","papers":2},{"task":"/task/document-classification","name":"Document Classification","papers":1},{"task":null,"name":"GPU","papers":1},{"task":"/task/knowledge-distillation","name":"Knowledge Distillation","papers":1},{"task":"/task/machine-translation","name":"Machine Translation","papers":1},{"task":"/task/model-compression","name":"Model Compression","papers":1},{"task":"/task/question-answering","name":"Question Answering","papers":1},{"task":"/task/text-classification","name":"Text Classification","papers":1},{"task":"/task/text-generation","name":"Text Generation","papers":1},{"task":"/task/transfer-learning","name":"Transfer Learning","papers":1},{"task":"/task/parameter-efficient-fine-tuning","name":"parameter-efficient fine-tuning","papers":1},{"task":"/task/text-classification-1","name":"text-classification","papers":1}],"tasks_shown":12,"n_tasks":12,"usage_by_year":[{"year":"2020","papers":1},{"year":"2021","papers":1},{"year":"2023","papers":2},{"year":"2024","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/movement-pruning"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}