{"url":"/task/model-compression","name":"Model Compression","slug":"model-compression","description_markdown":"**Model Compression** is an actively pursued area of research over the last few years with the goal of deploying state-of-the-art deep networks in low-power and resource limited devices without significant drop in accuracy. Parameter pruning, low-rank factorization and weight quantization are some of the proposed methods to compress the size of deep networks.\n\n\n<span class=\"description-source\">Source: [KD-MRI: A knowledge distillation framework for image reconstruction and image restoration in MRI workflow ](https://arxiv.org/abs/2004.05319)</span>","categories":[{"name":"Methodology","url":"/area/methodology"},{"name":"Miscellaneous","url":"/area/miscellaneous"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":1356,"papers_with_code":440,"benchmarks":2,"benchmark_tables_in_archive":2,"benchmark_tables_shown":2,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":3,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/model-compression-on-imagenet","slug":"model-compression-on-imagenet","dataset":"ImageNet","dataset_url":"/dataset/imagenet","rows_in_archive":12,"metrics":["Top-1"],"first_row_in_archive_order":{"model":"ADLIK-MO-ResNet50+W4A4","paper_title":"Learned Step Size Quantization","paper_url":"/paper/learned-step-size-quantization","paper_date":"2019-02-21","arxiv_id":"1902.08153","code_links":[{"title":"zhutmost/lsq-net","url":"https://github.com/zhutmost/lsq-net"},{"title":"hustzxd/LSQuantization","url":"https://github.com/hustzxd/LSQuantization"},{"title":"ZouJiu1/LSQplus","url":"https://github.com/ZouJiu1/LSQplus"},{"title":"Adlik/model_optimizer","url":"https://github.com/Adlik/model_optimizer"},{"title":"DeadAt0m/LSQFakeQuantize-PyTorch","url":"https://github.com/DeadAt0m/LSQFakeQuantize-PyTorch"},{"title":"DeadAt0m/LSQ-PyTorch","url":"https://github.com/DeadAt0m/LSQ-PyTorch"},{"title":"Shunli-Wang/Tiny-YOLO-LSQ","url":"https://github.com/Shunli-Wang/Tiny-YOLO-LSQ"},{"title":"Kelvinyu1117/LSQ-implementation","url":"https://github.com/Kelvinyu1117/LSQ-implementation"},{"title":"jiyoonkm/columnquant","url":"https://github.com/jiyoonkm/columnquant"}],"syntology":{"n":23,"n_ran":7,"n_unverified":16,"n_pointer_only":6}}},{"leaderboard":"/sota/model-compression-on-qnli","slug":"model-compression-on-qnli","dataset":"QNLI","dataset_url":"/dataset/qnli","rows_in_archive":2,"metrics":["Accuracy"],"first_row_in_archive_order":{"model":"MobileBERT + 2bit-1dim model compression using DKM","paper_title":"R2 Loss: Range Restriction Loss for Model Compression and Quantization","paper_url":"/paper/r-2-range-regularization-for-model","paper_date":"2023-03-14","arxiv_id":"2303.08253","code_links":[],"syntology":null}}],"datasets":[{"url":"/dataset/imagenet","name":"ImageNet","full_name":"","num_papers_in_archive":15430},{"url":"/dataset/glue","name":"GLUE","full_name":"General Language Understanding Evaluation benchmark","num_papers_in_archive":3197},{"url":"/dataset/qnli","name":"QNLI","full_name":"Question-answering NLI","num_papers_in_archive":1234}],"subtasks":[{"url":"/task/neural-network-compression","name":"Neural Network Compression"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":440,"tagged_in_all":1356,"items":[{"url":"/paper/squeezenet-alexnet-level-accuracy-with-50x","title":"SqueezeNet: AlexNet-level accuracy with 50x fewer parameters and <0.5MB model size","date":"2016-02-24","arxiv_id":"1602.07360","repositories_listed":59,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/well-read-students-learn-better-the-impact-of","title":"Well-Read Students Learn Better: On the Importance of Pre-training Compact Models","date":"2019-08-23","arxiv_id":"1908.08962","repositories_listed":40,"syntology":{"n":32,"n_ran":6,"n_unverified":26,"n_pointer_only":0}},{"url":"/paper/gptq-accurate-post-training-quantization-for","title":"GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers","date":"2022-10-31","arxiv_id":"2210.17323","repositories_listed":17,"syntology":{"n":15,"n_ran":5,"n_unverified":10,"n_pointer_only":1}},{"url":"/paper/amc-automl-for-model-compression-and","title":"AMC: AutoML for Model Compression and Acceleration on Mobile Devices","date":"2018-02-10","arxiv_id":"1802.03494","repositories_listed":12,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":1}},{"url":"/paper/learned-step-size-quantization","title":"Learned Step Size Quantization","date":"2019-02-21","arxiv_id":"1902.08153","repositories_listed":9,"syntology":{"n":23,"n_ran":7,"n_unverified":16,"n_pointer_only":6}},{"url":"/paper/the-state-of-sparsity-in-deep-neural-networks","title":"The State of Sparsity in Deep Neural Networks","date":"2019-02-25","arxiv_id":"1902.09574","repositories_listed":6,"syntology":null},{"url":"/paper/patient-knowledge-distillation-for-bert-model","title":"Patient Knowledge Distillation for BERT Model Compression","date":"2019-08-25","arxiv_id":"1908.09355","repositories_listed":5,"syntology":{"n":27,"n_ran":9,"n_unverified":18,"n_pointer_only":27}},{"url":"/paper/model-compression-via-distillation-and","title":"Model compression via distillation and quantization","date":"2018-02-15","arxiv_id":"1802.05668","repositories_listed":5,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/ternary-weight-networks","title":"Ternary Weight Networks","date":"2016-05-16","arxiv_id":"1605.04711","repositories_listed":5,"syntology":null},{"url":"/paper/sharpness-aware-quantization-for-deep-neural","title":"Sharpness-aware Quantization for Deep Neural Networks","date":"2021-11-24","arxiv_id":"2111.12273","repositories_listed":4,"syntology":{"n":11,"n_ran":0,"n_unverified":11,"n_pointer_only":0}},{"url":"/paper/lightspeech-lightweight-and-fast-text-to","title":"LightSpeech: Lightweight and Fast Text to Speech with Neural Architecture Search","date":"2021-02-08","arxiv_id":"2102.04040","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/training-with-quantization-noise-for-extreme","title":"Training with Quantization Noise for Extreme Model Compression","date":"2020-04-15","arxiv_id":"2004.07320","repositories_listed":4,"syntology":null},{"url":"/paper/contrastive-representation-distillation-1","title":"Contrastive Representation Distillation","date":"2019-10-23","arxiv_id":"1910.10699","repositories_listed":4,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/global-sparse-momentum-sgd-for-pruning-very","title":"Global Sparse Momentum SGD for Pruning Very Deep Neural Networks","date":"2019-09-27","arxiv_id":"1909.12778","repositories_listed":4,"syntology":{"n":5,"n_ran":1,"n_unverified":4,"n_pointer_only":0}},{"url":"/paper/to-prune-or-not-to-prune-exploring-the","title":"To prune, or not to prune: exploring the efficacy of pruning for model compression","date":"2017-10-05","arxiv_id":"1710.01878","repositories_listed":4,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/deepspeed-moe-advancing-mixture-of-experts","title":"DeepSpeed-MoE: Advancing Mixture-of-Experts Inference and Training to Power Next-Generation AI Scale","date":"2022-01-14","arxiv_id":"2201.05596","repositories_listed":3,"syntology":null},{"url":"/paper/zeroq-a-novel-zero-shot-quantization","title":"ZeroQ: A Novel Zero Shot Quantization Framework","date":"2020-01-01","arxiv_id":"2001.00281","repositories_listed":3,"syntology":{"n":19,"n_ran":4,"n_unverified":15,"n_pointer_only":0}},{"url":"/paper/data-free-adversarial-distillation","title":"Data-Free Adversarial Distillation","date":"2019-12-23","arxiv_id":"1912.11006","repositories_listed":3,"syntology":{"n":2,"n_ran":1,"n_unverified":1,"n_pointer_only":2}},{"url":"/paper/microexpnet-an-extremely-small-and-fast-model","title":"MicroExpNet: An Extremely Small and Fast Model For Expression Recognition From Face Images","date":"2017-11-19","arxiv_id":"1711.07011","repositories_listed":3,"syntology":null},{"url":"/paper/actor-mimic-deep-multitask-and-transfer","title":"Actor-Mimic: Deep Multitask and Transfer Reinforcement Learning","date":"2015-11-19","arxiv_id":"1511.06342","repositories_listed":3,"syntology":null},{"url":"/paper/localize-and-stitch-efficient-model-merging","title":"Localize-and-Stitch: Efficient Model Merging via Sparse Task Arithmetic","date":"2024-08-24","arxiv_id":"2408.13656","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/adversarial-fine-tuning-of-compressed-neural","title":"Adversarial Fine-tuning of Compressed Neural Networks for Joint Improvement of Robustness and Efficiency","date":"2024-03-14","arxiv_id":"2403.09441","repositories_listed":2,"syntology":null},{"url":"/paper/llm-inference-unveiled-survey-and-roofline","title":"LLM Inference Unveiled: Survey and Roofline Model Insights","date":"2024-02-26","arxiv_id":"2402.16363","repositories_listed":2,"syntology":{"n":10,"n_ran":10,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/safety-and-performance-why-not-both-bi-1","title":"Safety and Performance, Why Not Both? Bi-Objective Optimized Model Compression against Heterogeneous Attacks Toward AI Software Deployment","date":"2024-01-02","arxiv_id":"2401.00996","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_unverified":1,"n_pointer_only":3}},{"url":"/paper/addressing-membership-inference-attack-in","title":"Privacy and Accuracy Implications of Model Complexity and Integration in Heterogeneous Federated Learning","date":"2023-11-29","arxiv_id":"2311.17750","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_unverified":1,"n_pointer_only":5}},{"url":"/paper/lxmert-model-compression-for-visual-question","title":"LXMERT Model Compression for Visual Question Answering","date":"2023-10-23","arxiv_id":"2310.15325","repositories_listed":2,"syntology":null},{"url":"/paper/reprogramming-under-constraints-revisiting","title":"Uncovering the Hidden Cost of Model Compression","date":"2023-08-29","arxiv_id":"2308.14969","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/omniquant-omnidirectionally-calibrated","title":"OmniQuant: Omnidirectionally Calibrated Quantization for Large Language Models","date":"2023-08-25","arxiv_id":"2308.13137","repositories_listed":2,"syntology":{"n":16,"n_ran":9,"n_unverified":7,"n_pointer_only":0}},{"url":"/paper/variation-aware-vision-transformer","title":"Quantization Variation: A New Perspective on Training Transformers with Low-Bit Precision","date":"2023-07-01","arxiv_id":"2307.00331","repositories_listed":2,"syntology":null},{"url":"/paper/performance-aware-approximation-of-global","title":"Performance-aware Approximation of Global Channel Pruning for Multitask CNNs","date":"2023-03-21","arxiv_id":"2303.11923","repositories_listed":2,"syntology":null}],"syntology_records":20,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}