{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/model-compression/papers/11","list_of":"/task/model-compression","task":"Model Compression","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":11,"pages_in_order":14,"rows_per_page":100,"rows":[1001,1100],"of":1356,"counts":{"archive_papers_tagged":1356,"with_a_code_link":440,"where_syntology_ran_a_sample":119,"not_listed_spam_title":0,"listed":1356,"listed_where_code_ran":119,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":97,"every_run_a_failure_of_syntologys_instrument":22,"listed_with_a_run_with_no_instrument_failure":97,"listed_every_run_a_failure_of_syntologys_instrument":22,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/model-compression","prev":"/task/model-compression/papers/10","next":"/task/model-compression/papers/12","papers":[{"url":null,"slug":"lipschitz-continuity-guided-knowledge","title":"Lipschitz Continuity Guided Knowledge Distillation","date":"2021-08-29","arxiv_id":"2108.12905","repositories_listed":0,"syntology":null},{"url":null,"slug":"dkm-differentiable-k-means-clustering-layer","title":"DKM: Differentiable K-Means Clustering Layer for Neural Network Compression","date":"2021-08-28","arxiv_id":"2108.12659","repositories_listed":0,"syntology":null},{"url":null,"slug":"yanmtt-yet-another-neural-machine-translation","title":"YANMTT: Yet Another Neural Machine Translation Toolkit","date":"2021-08-25","arxiv_id":"2108.11126","repositories_listed":0,"syntology":null},{"url":null,"slug":"small-object-detection-based-on-modified-fssd","title":"Small Object Detection Based on Modified FSSD and Model Compression","date":"2021-08-24","arxiv_id":"2108.10503","repositories_listed":0,"syntology":null},{"url":null,"slug":"scaling-laws-for-deep-learning","title":"Scaling Laws for Deep Learning","date":"2021-08-17","arxiv_id":"2108.07686","repositories_listed":0,"syntology":null},{"url":null,"slug":"preventing-catastrophic-forgetting-and","title":"Preventing Catastrophic Forgetting and Distribution Mismatch in Knowledge Distillation via Synthetic Data","date":"2021-08-11","arxiv_id":"2108.05698","repositories_listed":0,"syntology":null},{"url":null,"slug":"random-offset-block-embedding-array-robe-for","title":"Random Offset Block Embedding Array (ROBE) for CriteoTB Benchmark MLPerf DLRM Model : 1000$\\times$ Compression and 3.1$\\times$ Faster Inference","date":"2021-08-04","arxiv_id":"2108.02191","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-a-neural-diff-for-speech-models","title":"Learning a Neural Diff for Speech Models","date":"2021-08-03","arxiv_id":"2108.01561","repositories_listed":0,"syntology":null},{"url":null,"slug":"quped-quantized-personalization-via","title":"QuPeD: Quantized Personalization via Distillation with Applications to Federated Learning","date":"2021-07-29","arxiv_id":"2107.13892","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-efficient-tensor-decomposition-based-1","title":"Towards Efficient Tensor Decomposition-Based DNN Model Compression with Optimization Framework","date":"2021-07-26","arxiv_id":"2107.12422","repositories_listed":0,"syntology":null},{"url":null,"slug":"pruning-ternary-quantization","title":"Pruning Ternary Quantization","date":"2021-07-23","arxiv_id":"2107.10998","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-new-clustering-based-technique-for-the","title":"A New Clustering-Based Technique for the Acceleration of Deep Convolutional Networks","date":"2021-07-19","arxiv_id":"2107.09095","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-deep-neural-networks-for","title":"Accelerating deep neural networks for efficient scene understanding in automotive cyber-physical systems","date":"2021-07-19","arxiv_id":"2107.09101","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-action-recognition-on-heterogeneous","title":"Federated Action Recognition on Heterogeneous Embedded Devices","date":"2021-07-18","arxiv_id":"2107.12147","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-automated-u-net-based-tree-crown","title":"Efficient automated U-Net based tree crown delineation using UAV multi-spectral imagery on embedded devices","date":"2021-07-16","arxiv_id":"2107.07826","repositories_listed":0,"syntology":null},{"url":null,"slug":"weclick-weakly-supervised-video-semantic","title":"WeClick: Weakly-Supervised Video Semantic Segmentation with Click Annotations","date":"2021-07-07","arxiv_id":"2107.03088","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-accurate-human-activity-recognition","title":"A Light-weight Deep Human Activity Recognition Algorithm Using Multi-knowledge Distillation","date":"2021-07-06","arxiv_id":"2107.07331","repositories_listed":0,"syntology":null},{"url":null,"slug":"investigation-of-practical-aspects-of-single","title":"Investigation of Practical Aspects of Single Channel Speech Separation for ASR","date":"2021-07-05","arxiv_id":"2107.01922","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-lottery-ticket-hypothesis-framework-for-low","title":"A Lottery Ticket Hypothesis Framework for Low-Complexity Device-Robust Neural Acoustic Scene Classification","date":"2021-07-03","arxiv_id":"2107.01461","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-teacher-forcing-network-for-semi","title":"Scalable Teacher Forcing Network for Semi-Supervised Large Scale Data Streams","date":"2021-06-26","arxiv_id":"2107.02943","repositories_listed":0,"syntology":null},{"url":null,"slug":"pqk-model-compression-via-pruning","title":"PQK: Model Compression via Pruning, Quantization, and Knowledge Distillation","date":"2021-06-25","arxiv_id":"2106.14681","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-free-knowledge-distillation-for-image","title":"Data-Free Knowledge Distillation for Image Super-Resolution","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"minimally-invasive-surgery-for-sparse-neural","title":"Minimally Invasive Surgery for Sparse Neural Networks in Contrastive Manner","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneous-training-of-partially-masked","title":"Masked Training of Neural Networks with Partial Gradients","date":"2021-06-16","arxiv_id":"2106.08895","repositories_listed":0,"syntology":null},{"url":null,"slug":"topology-distillation-for-recommender-system","title":"Topology Distillation for Recommender System","date":"2021-06-16","arxiv_id":"2106.08700","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-efficient-knowledge-distillation-for","title":"Energy-efficient Knowledge Distillation for Spiking Neural Networks","date":"2021-06-14","arxiv_id":"2106.07172","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-dynamic-pruning-for-non-iid","title":"Heterogeneous Federated Learning using Dynamic Model Pruning and Adaptive Gradient","date":"2021-06-13","arxiv_id":"2106.06921","repositories_listed":0,"syntology":null},{"url":null,"slug":"fednilm-applying-federated-learning-to-nilm","title":"FedNILM: Applying Federated Learning to NILM Applications at the Edge","date":"2021-06-07","arxiv_id":"2106.07751","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-flow-regularization-improving","title":"Feature Flow Regularization: Improving Structured Sparsity in Deep Neural Networks","date":"2021-06-05","arxiv_id":"2106.02914","repositories_listed":0,"syntology":null},{"url":null,"slug":"fednl-making-newton-type-methods-applicable","title":"FedNL: Making Newton-Type Methods Applicable to Federated Learning","date":"2021-06-05","arxiv_id":"2106.02969","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-efficient-model-compression-and","title":"Energy-Efficient Model Compression and Splitting for Collaborative Inference Over Time-Varying Channels","date":"2021-06-02","arxiv_id":"2106.00995","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-teacher-is-enough-pre-trained-language","title":"One Teacher is Enough? Pre-trained Language Model Distillation from Multiple Teachers","date":"2021-06-02","arxiv_id":"2106.01023","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-attention-redundancy-a-comprehensive-study","title":"On Attention Redundancy: A Comprehensive Study","date":"2021-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"nas-bert-task-agnostic-and-adaptive-size-bert","title":"NAS-BERT: Task-Agnostic and Adaptive-Size BERT Compression with Neural Architecture Search","date":"2021-05-30","arxiv_id":"2105.14444","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-efficient-full-8-bit-integer-dnn","title":"Towards Efficient Full 8-bit Integer DNN Online Training on Resource-limited Devices without Batch Normalization","date":"2021-05-27","arxiv_id":"2105.13890","repositories_listed":0,"syntology":null},{"url":null,"slug":"differentiable-sparsification-for-deep-neural-1","title":"Differentiable Sparsification for Deep Neural Networks","date":"2021-05-21","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"model-compression","title":"Model Compression","date":"2021-05-20","arxiv_id":"2105.10059","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-explain-neural-networks-a-perspective","title":"How to Explain Neural Networks: an Approximation Perspective","date":"2021-05-17","arxiv_id":"2105.07831","repositories_listed":0,"syntology":null},{"url":null,"slug":"3u-edgeai-ultra-low-memory-training-ultra-low","title":"3U-EdgeAI: Ultra-Low Memory Training, Ultra-Low BitwidthQuantization, and Ultra-Low Latency Acceleration","date":"2021-05-11","arxiv_id":"2105.06250","repositories_listed":0,"syntology":null},{"url":null,"slug":"test-time-adaptation-toward-personalized","title":"Test-Time Adaptation Toward Personalized Speech Enhancement: Zero-Shot Learning with Knowledge Distillation","date":"2021-05-08","arxiv_id":"2105.03544","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-3d-scene-compression-via-model","title":"Neural 3D Scene Compression via Model Compression","date":"2021-05-07","arxiv_id":"2105.03120","repositories_listed":0,"syntology":null},{"url":null,"slug":"modulating-regularization-frequency-for","title":"Modulating Regularization Frequency for Efficient Compression-Aware Model Training","date":"2021-05-05","arxiv_id":"2105.01875","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequential-encryption-of-sparse-neural","title":"Encoding Weights of Irregular Sparsity for Fixed-to-Fixed Model Compression","date":"2021-05-05","arxiv_id":"2105.01869","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-for-swedish-ner-models","title":"Knowledge Distillation for Swedish NER models: A Search for Performance and Efficiency","date":"2021-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-adversarial-robustness-of-quantized","title":"On the Adversarial Robustness of Quantized Neural Networks","date":"2021-05-01","arxiv_id":"2105.00227","repositories_listed":0,"syntology":null},{"url":null,"slug":"spirit-distillation-a-model-compression","title":"Spirit Distillation: A Model Compression Method with Multi-domain Knowledge Transfer","date":"2021-04-29","arxiv_id":"2104.14696","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatio-temporal-pruning-and-quantization-for","title":"Spatio-Temporal Pruning and Quantization for Low-latency Spiking Neural Networks","date":"2021-04-26","arxiv_id":"2104.12528","repositories_listed":0,"syntology":null},{"url":null,"slug":"compact-cnn-structure-learning-by-knowledge","title":"Compact CNN Structure Learning by Knowledge Distillation","date":"2021-04-19","arxiv_id":"2104.09191","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-discriminator-adversarial-distillation","title":"Dual Discriminator Adversarial Distillation for Data-free Model Compression","date":"2021-04-12","arxiv_id":"2104.05382","repositories_listed":0,"syntology":null},{"url":null,"slug":"reversible-watermarking-in-deep-convolutional","title":"Reversible Watermarking in Deep Convolutional Neural Networks for Integrity Authentication","date":"2021-04-09","arxiv_id":"2104.04268","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-learning-for-personalized","title":"Efficient Personalized Speech Enhancement through Self-Supervised Learning","date":"2021-04-05","arxiv_id":"2104.02017","repositories_listed":0,"syntology":null},{"url":null,"slug":"tight-compression-compressing-cnn-through","title":"Tight Compression: Compressing CNN Through Fine-Grained Pruning and Weight Permutation for Efficient Implementation","date":"2021-04-03","arxiv_id":"2104.01303","repositories_listed":0,"syntology":null},{"url":null,"slug":"shrinking-bigfoot-reducing-wav2vec-2-0","title":"Shrinking Bigfoot: Reducing wav2vec 2.0 footprint","date":"2021-03-29","arxiv_id":"2103.15760","repositories_listed":0,"syntology":null},{"url":null,"slug":"prototype-based-personalized-pruning","title":"Prototype-based Personalized Pruning","date":"2021-03-25","arxiv_id":"2103.15564","repositories_listed":0,"syntology":null},{"url":null,"slug":"compacting-deep-neural-networks-for-internet","title":"Compacting Deep Neural Networks for Internet of Things: Methods and Applications","date":"2021-03-20","arxiv_id":"2103.11083","repositories_listed":0,"syntology":null},{"url":null,"slug":"mwq-multiscale-wavelet-quantized-neural","title":"MWQ: Multiscale Wavelet Quantized Neural Networks","date":"2021-03-09","arxiv_id":"2103.05363","repositories_listed":0,"syntology":null},{"url":null,"slug":"formalizing-generalization-and-robustness-of-1","title":"Formalizing Generalization and Robustness of Neural Networks to Weight Perturbations","date":"2021-03-03","arxiv_id":"2103.02200","repositories_listed":0,"syntology":null},{"url":null,"slug":"pursuhint-in-search-of-informative-hint","title":"PURSUhInT: In Search of Informative Hint Points Based on Layer Clustering for Knowledge Distillation","date":"2021-02-26","arxiv_id":"2103.00053","repositories_listed":0,"syntology":null},{"url":null,"slug":"lottery-ticket-implies-accuracy-degradation","title":"Lottery Ticket Preserves Weight Correlation: Is It Desirable or Not?","date":"2021-02-19","arxiv_id":"2102.11068","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-network-compression-for-noisy-storage","title":"Neural Network Compression for Noisy Storage Devices","date":"2021-02-15","arxiv_id":"2102.07725","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustness-in-compressed-neural-networks-for","title":"Robustness in Compressed Neural Networks for Object Detection","date":"2021-02-10","arxiv_id":"2102.05509","repositories_listed":0,"syntology":null},{"url":null,"slug":"it-s-always-personal-using-early-exits-for","title":"It's always personal: Using Early Exits for Efficient On-Device CNN Personalisation","date":"2021-02-02","arxiv_id":"2102.01393","repositories_listed":0,"syntology":null},{"url":null,"slug":"aacp-model-compression-by-accurate-and","title":"AACP: Model Compression by Accurate and Automatic Channel Pruning","date":"2021-01-31","arxiv_id":"2102.00390","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-model-compression-based-on-the-training","title":"Deep Model Compression based on the Training History","date":"2021-01-30","arxiv_id":"2102.00160","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaspring-context-adaptive-and-runtime","title":"AdaSpring: Context-adaptive and Runtime-evolutionary Deep Model Compression for Mobile Applications","date":"2021-01-28","arxiv_id":"2101.11800","repositories_listed":0,"syntology":null},{"url":null,"slug":"differential-privacy-meets-federated-learning","title":"Differential Privacy Meets Federated Learning under Communication Constraints","date":"2021-01-28","arxiv_id":"2101.12240","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-teacher-student-learning-via","title":"Collaborative Teacher-Student Learning via Multiple Knowledge Transfer","date":"2021-01-21","arxiv_id":"2101.08471","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-compression-of-neural-networks-for-fault","title":"Deep Compression of Neural Networks for Fault Detection on Tennessee Eastman Chemical Processes","date":"2021-01-18","arxiv_id":"2101.06993","repositories_listed":0,"syntology":null},{"url":null,"slug":"abs-automatic-bit-sharing-for-model-1","title":"Single-path Bit Sharing for Automatic Loss-aware Model Compression","date":"2021-01-13","arxiv_id":"2101.04935","repositories_listed":0,"syntology":null},{"url":null,"slug":"activation-density-based-mixed-precision","title":"Activation Density based Mixed-Precision Quantization for Energy Efficient Neural Networks","date":"2021-01-12","arxiv_id":"2101.04354","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarially-robust-and-explainable-model","title":"Adversarially Robust and Explainable Model Compression with On-Device Personalization for Text Classification","date":"2021-01-10","arxiv_id":"2101.05624","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-device-document-classification-using","title":"On-Device Document Classification using multimodal features","date":"2021-01-06","arxiv_id":"2101.01880","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-half-space-stochastic-projected-gradient","title":"A Half-Space Stochastic Projected Gradient Method for Group Sparsity Regularization","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"block-skim-transformer-for-efficient-question","title":"Block Skim Transformer for Efficient Question Answering","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"can-students-outperform-teachers-in-knowledge","title":"Can Students Outperform Teachers in Knowledge Distillation based Model Compression?","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-probabilistic-pruning-training-sparse","title":"Dynamic Probabilistic Pruning: Training sparse networks based on stochastic and dynamic masking","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploration-and-estimation-for-model","title":"Exploration and Estimation for Model Compression","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-via-softmax-regression","title":"Knowledge distillation via softmax regression representation learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"model-compression-via-hyper-structure-network","title":"Model Compression via Hyper-Structure Network","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"post-training-weighted-quantization-of-neural","title":"Post-Training Weighted Quantization of Neural Networks for Language Models","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"task-agnostic-and-adaptive-size-bert","title":"Task-Agnostic and Adaptive-Size BERT Compression","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"twindnn-a-tale-of-two-deep-neural-networks","title":"TwinDNN: A Tale of Two Deep Neural Networks","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-zero-shot-knowledge-distillation-for","title":"Towards Zero-Shot Knowledge Distillation for Natural Language Processing","date":"2020-12-31","arxiv_id":"2012.15495","repositories_listed":0,"syntology":null},{"url":"/paper/a-surrogate-lagrangian-relaxation-based-model","slug":"a-surrogate-lagrangian-relaxation-based-model","title":"Enabling Retrain-free Deep Neural Network Pruning using Surrogate Lagrangian Relaxation","date":"2020-12-18","arxiv_id":"2012.10079","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-optimal-neural-networks-rapid","title":"Distilling Optimal Neural Networks: Rapid Search in Diverse Spaces","date":"2020-12-16","arxiv_id":"2012.08859","repositories_listed":0,"syntology":null},{"url":null,"slug":"provable-benefits-of-overparameterization-in","title":"Provable Benefits of Overparameterization in Model Compression: From Double Descent to Pruning Neural Networks","date":"2020-12-16","arxiv_id":"2012.08749","repositories_listed":0,"syntology":null},{"url":"/paper/wasserstein-contrastive-representation","slug":"wasserstein-contrastive-representation","title":"Wasserstein Contrastive Representation Distillation","date":"2020-12-15","arxiv_id":"2012.08674","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-multi-teacher-selection-for","title":"Reinforced Multi-Teacher Selection for Knowledge Distillation","date":"2020-12-11","arxiv_id":"2012.06048","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-generative-data-free-distillation","title":"Large-Scale Generative Data-Free Distillation","date":"2020-12-10","arxiv_id":"2012.05578","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-lightweight-neural-network-for-inferring","title":"Inferring ECG from PPG for Continuous Cardiac Monitoring Using Lightweight Neural Network","date":"2020-12-09","arxiv_id":"2012.04949","repositories_listed":0,"syntology":null},{"url":null,"slug":"mix-and-match-a-novel-fpga-centric-deep","title":"Mix and Match: A Novel FPGA-Centric Deep Neural Network Quantization Framework","date":"2020-12-08","arxiv_id":"2012.04240","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-compression-using-optimal-transport","title":"Model Compression Using Optimal Transport","date":"2020-12-07","arxiv_id":"2012.03907","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-head-knowledge-distillation-for-model","title":"Multi-head Knowledge Distillation for Model Compression","date":"2020-12-05","arxiv_id":"2012.02911","repositories_listed":0,"syntology":null},{"url":null,"slug":"6-7ms-on-mobile-with-over-78-imagenet","title":"NPAS: A Compiler-aware Framework of Unified Network Pruning and Architecture Search for Beyond Real-Time Mobile Acceleration","date":"2020-12-01","arxiv_id":"2012.00596","repositories_listed":0,"syntology":null},{"url":null,"slug":"compressing-pre-trained-language-models-by","title":"Compressing Pre-trained Language Models by Matrix Decomposition","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"edge-deep-learning-for-neural-implants","title":"Edge Deep Learning for Neural Implants","date":"2020-12-01","arxiv_id":"2012.00307","repositories_listed":0,"syntology":null},{"url":null,"slug":"reverse-engineering-recurrent-neural-network","title":"Reverse-engineering recurrent neural network solutions to a hierarchical inference task for mice","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-generative-adversarial-1","title":"Self-Supervised Generative Adversarial Compression","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"a-selective-survey-on-versatile-knowledge","title":"A Selective Survey on Versatile Knowledge Distillation Paradigm for Neural Network Models","date":"2020-11-30","arxiv_id":"2011.14554","repositories_listed":0,"syntology":null},{"url":null,"slug":"extreme-model-compression-for-on-device","title":"Extreme Model Compression for On-device Natural Language Understanding","date":"2020-11-30","arxiv_id":"2012.00124","repositories_listed":0,"syntology":null}],"record_sha256":"67cec3180bb845f5816237122c81eefc7476eb3142dd5cecf34a28f1ab9cbf2c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}