{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/knowledge-distillation/papers/42","list_of":"/task/knowledge-distillation","task":"Knowledge Distillation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":42,"pages_in_order":43,"rows_per_page":100,"rows":[4101,4200],"of":4240,"counts":{"archive_papers_tagged":4240,"with_a_code_link":1740,"where_syntology_ran_a_sample":451,"not_listed_spam_title":0,"listed":4240,"listed_where_code_ran":451,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":380,"every_run_a_failure_of_syntologys_instrument":71,"listed_with_a_run_with_no_instrument_failure":380,"listed_every_run_a_failure_of_syntologys_instrument":71,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/knowledge-distillation","prev":"/task/knowledge-distillation/papers/41","next":"/task/knowledge-distillation/papers/43","papers":[{"url":null,"slug":"model-compression-with-two-stage-multi","title":"Model Compression with Two-stage Multi-teacher Knowledge Distillation for Web Question Answering System","date":"2019-10-18","arxiv_id":"1910.08381","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-generalized-and-robust-method-towards","title":"A Generalized and Robust Method Towards Practical Gaze Estimation on Smart Phone","date":"2019-10-16","arxiv_id":"1910.07331","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-generalization-and-robustness-with","title":"Noise as a Resource for Learning in Knowledge Distillation","date":"2019-10-11","arxiv_id":"1910.05057","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-modal-knowledge-distillation-for-action","title":"Cross-modal knowledge distillation for action recognition","date":"2019-10-10","arxiv_id":"1910.04641","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-from-internal","title":"Knowledge Distillation from Internal Representations","date":"2019-10-08","arxiv_id":"1910.03723","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-transformers-into-simple-neural","title":"Distilling BERT into Simple Neural Networks with Unlabeled Transfer Data","date":"2019-10-04","arxiv_id":"1910.01769","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-efficacy-of-knowledge-distillation","title":"On the Efficacy of Knowledge Distillation","date":"2019-10-03","arxiv_id":"1910.01348","repositories_listed":0,"syntology":null},{"url":null,"slug":"antman-sparse-low-rank-compression-to-1","title":"AntMan: Sparse Low-Rank Compression to Accelerate RNN inference","date":"2019-10-02","arxiv_id":"1910.01740","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilled-embedding-non-linear-embedding","title":"Improving Word Embedding Factorization for Compression Using Distilled Nonlinear Neural Decomposition","date":"2019-10-02","arxiv_id":"1910.06720","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bayesian-optimization-framework-for-neural","title":"A Bayesian Optimization Framework for Neural Network Compression","date":"2019-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-inter-agent-knowledge","title":"Collaborative Inter-agent Knowledge Distillation for Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"distilled-embedding-non-linear-embedding-1","title":"Distilled embedding: non-linear embedding factorization using knowledge distillation","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"extreme-language-model-compression-with-1","title":"Extremely Small BERT Models from Mixed-Vocabulary Training","date":"2019-09-25","arxiv_id":"1909.11687","repositories_listed":0,"syntology":null},{"url":null,"slug":"proactive-sequence-generator-via-knowledge","title":"Proactive Sequence Generator via Knowledge Acquisition","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"self-knowledge-distillation-adversarial","title":"SELF-KNOWLEDGE DISTILLATION ADVERSARIAL ATTACK","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"xd-cross-lingual-knowledge-distillation-for","title":"XD: Cross-lingual Knowledge Distillation for Polyglot Sentence Embeddings","date":"2019-09-25","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"feed-feature-level-ensemble-for-knowledge","title":"FEED: Feature-level Ensemble for Knowledge Distillation","date":"2019-09-24","arxiv_id":"1909.10754","repositories_listed":0,"syntology":null},{"url":null,"slug":"technical-report-on-conversational-question","title":"Technical report on Conversational Question Answering","date":"2019-09-24","arxiv_id":"1909.10772","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-lightweight-pedestrian-detector-with","title":"Learning Lightweight Pedestrian Detector with Hierarchical Knowledge Distillation","date":"2019-09-20","arxiv_id":"1909.09325","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-transformer-decoding-via-a","title":"Accelerating Transformer Decoding via a Hybrid of Self-attention and Recurrent Neural Network","date":"2019-09-05","arxiv_id":"1909.02279","repositories_listed":0,"syntology":null},{"url":null,"slug":"empirical-analysis-of-knowledge-distillation","title":"Knowledge distillation for optimization of quantized deep neural networks","date":"2019-09-04","arxiv_id":"1909.01688","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-sensor-hallucination-via-knowledge","title":"Online Sensor Hallucination via Knowledge Distillation for Multimodal Image Classification","date":"2019-08-28","arxiv_id":"1908.10559","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-based-knowledge-distillation-for","title":"Adversarial-Based Knowledge Distillation for Multi-Model Ensemble and Noisy Data Refinement","date":"2019-08-22","arxiv_id":"1908.08520","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-graph-distillation-for-low-resource","title":"Language Graph Distillation for Low-Resource Machine Translation","date":"2019-08-17","arxiv_id":"1908.06258","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-for-semi-supervised","title":"Knowledge distillation for semi-supervised domain adaptation","date":"2019-08-16","arxiv_id":"1908.07355","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-regularization-of-labels","title":"Adaptive Regularization of Labels","date":"2019-08-15","arxiv_id":"1908.05474","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-training-of-convolutional-neural","title":"Effective Training of Convolutional Neural Networks with Low-bitwidth Weights and Activations","date":"2019-08-10","arxiv_id":"1908.04680","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-isomorphism-between-neural-networks","title":"Knowledge Consistency between Neural Networks and Beyond","date":"2019-08-05","arxiv_id":"1908.01581","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-knowledge-distillation-in-natural","title":"Self-Knowledge Distillation in Natural Language Processing","date":"2019-08-02","arxiv_id":"1908.01851","repositories_listed":0,"syntology":null},{"url":null,"slug":"baidu-neural-machine-translation-systems-for","title":"Baidu Neural Machine Translation Systems for WMT19","date":"2019-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"gtcom-neural-machine-translation-systems-for","title":"GTCOM Neural Machine Translation Systems for WMT19","date":"2019-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"panlp-at-mediqa-2019-pre-trained-language","title":"PANLP at MEDIQA 2019: Pre-trained Language Models, Transfer Learning and Knowledge Distillation","date":"2019-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-niutrans-machine-translation-systems-for","title":"The NiuTrans Machine Translation Systems for WMT19","date":"2019-08-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"distill-to-label-weakly-supervised-instance","title":"Distill-to-Label: Weakly Supervised Instance Labeling Using Knowledge Distillation","date":"2019-07-26","arxiv_id":"1907.12926","repositories_listed":0,"syntology":null},{"url":null,"slug":"teacher-students-knowledge-distillation-for","title":"Distilled Siamese Networks for Visual Tracking","date":"2019-07-24","arxiv_id":"1907.10586","repositories_listed":0,"syntology":null},{"url":null,"slug":"lifelong-gan-continual-learning-for","title":"Lifelong GAN: Continual Learning for Conditional Image Generation","date":"2019-07-23","arxiv_id":"1907.10107","repositories_listed":0,"syntology":null},{"url":null,"slug":"learn-spelling-from-teachers-transferring","title":"Learn Spelling from Teachers: Transferring Knowledge from Language Models to Sequence-to-Sequence Speech Recognition","date":"2019-07-13","arxiv_id":"1907.06017","repositories_listed":0,"syntology":null},{"url":null,"slug":"compression-of-acoustic-event-detection-1","title":"Compression of Acoustic Event Detection Models With Quantized Distillation","date":"2019-07-01","arxiv_id":"1907.00873","repositories_listed":0,"syntology":null},{"url":null,"slug":"reconstructing-perceived-images-from-brain","title":"Reconstructing Perceived Images from Brain Activity by Visually-guided Cognitive Representation and Adversarial Learning","date":"2019-06-27","arxiv_id":"1906.12181","repositories_listed":0,"syntology":null},{"url":null,"slug":"essence-knowledge-distillation-for-speech","title":"Essence Knowledge Distillation for Speech Recognition","date":"2019-06-26","arxiv_id":"1906.10834","repositories_listed":0,"syntology":null},{"url":null,"slug":"gan-knowledge-distillation-for-one-stage","title":"GAN-Knowledge Distillation for one-stage Object Detection","date":"2019-06-20","arxiv_id":"1906.08467","repositories_listed":0,"syntology":null},{"url":null,"slug":"reconciling-utility-and-membership-privacy","title":"Membership Privacy for Machine Learning Models Through Knowledge Transfer","date":"2019-06-15","arxiv_id":"1906.06589","repositories_listed":0,"syntology":null},{"url":null,"slug":"divide-and-conquer-leveraging-intermediate","title":"Divide and Conquer: Leveraging Intermediate Feature Representations for Quantized Training of Neural Networks","date":"2019-06-14","arxiv_id":"1906.06033","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-syntax-aware-language-models-using","title":"Scalable Syntax-Aware Language Models Using Knowledge Distillation","date":"2019-06-14","arxiv_id":"1906.06438","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-evaluation-time-uncertainty","title":"Efficient Evaluation-Time Uncertainty Estimation by Improved Distillation","date":"2019-06-12","arxiv_id":"1906.05419","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-classifier-learning-based-on","title":"Incremental Classifier Learning Based on PEDCC-Loss and Cosine Distance","date":"2019-06-11","arxiv_id":"1906.04734","repositories_listed":0,"syntology":null},{"url":null,"slug":"private-deep-learning-with-teacher-ensembles","title":"Private Deep Learning with Teacher Ensembles","date":"2019-06-05","arxiv_id":"1906.02303","repositories_listed":0,"syntology":null},{"url":null,"slug":"190600619","title":"Deep Face Recognition Model Compression via Knowledge Transfer and Distillation","date":"2019-06-03","arxiv_id":"1906.00619","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-knowledge-distillation-from-complex","title":"On Knowledge distillation from complex networks for response prediction","date":"2019-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"online-distilling-from-checkpoints-for-neural","title":"Online Distilling from Checkpoints for Neural Machine Translation","date":"2019-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-resolution-face-recognition-via-prior","title":"Cross-Resolution Face Recognition via Prior-Aided Face Hallucination and Residual Knowledge Distillation","date":"2019-05-26","arxiv_id":"1905.10777","repositories_listed":0,"syntology":null},{"url":null,"slug":"creating-lightweight-object-detectors-with","title":"Creating Lightweight Object Detectors with Model Compression for Deployment on Edge Devices","date":"2019-05-06","arxiv_id":"1905.01787","repositories_listed":0,"syntology":null},{"url":null,"slug":"feed-feature-level-ensemble-effect-for","title":"FEED: Feature-level Ensemble Effect for knowledge Distillation","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-better-understanding-of-vector","title":"Towards a better understanding of Vector Quantized Autoencoders","date":"2019-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"semi-supervised-acoustic-event-detection","title":"Semi-supervised Acoustic Event Detection based on tri-training","date":"2019-04-29","arxiv_id":"1904.12926","repositories_listed":0,"syntology":null},{"url":null,"slug":"segmenting-the-future","title":"Segmenting the Future","date":"2019-04-24","arxiv_id":"1904.10666","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-compression-with-multi-task-knowledge","title":"Model Compression with Multi-Task Knowledge Distillation for Web-scale Question Answering System","date":"2019-04-21","arxiv_id":"1904.09636","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-speech-translation-with-knowledge","title":"End-to-End Speech Translation with Knowledge Distillation","date":"2019-04-17","arxiv_id":"1904.08075","repositories_listed":0,"syntology":null},{"url":null,"slug":"guiding-ctc-posterior-spike-timings-for","title":"Guiding CTC Posterior Spike Timings for Improved Posterior Fusion and Knowledge Distillation","date":"2019-04-17","arxiv_id":"1904.08311","repositories_listed":0,"syntology":null},{"url":null,"slug":"examining-the-mapping-functions-of-denoising","title":"Examining the Mapping Functions of Denoising Autoencoders in Singing Voice Separation","date":"2019-04-12","arxiv_id":"1904.06157","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-squeezed-adversarial-network","title":"Knowledge Squeezed Adversarial Network Compression","date":"2019-04-10","arxiv_id":"1904.05100","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatiotemporal-knowledge-distillation-for","title":"Spatiotemporal Knowledge Distillation for Efficient Estimation of Aerial Video Saliency","date":"2019-04-10","arxiv_id":"1904.04992","repositories_listed":0,"syntology":null},{"url":null,"slug":"back-to-the-future-knowledge-distillation-for","title":"Knowledge Distillation for Human Action Anticipation","date":"2019-04-09","arxiv_id":"1904.04868","repositories_listed":0,"syntology":null},{"url":null,"slug":"ultrafast-video-attention-prediction-with","title":"Ultrafast Video Attention Prediction with Coupled Knowledge Distillation","date":"2019-04-09","arxiv_id":"1904.04449","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-for-recurrent-neural","title":"Knowledge Distillation For Recurrent Neural Network Language Modeling With Trust Regularization","date":"2019-04-08","arxiv_id":"1904.04163","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-term-vehicle-localization-by-recursive","title":"Long-Term Vehicle Localization by Recursive Knowledge Distillation","date":"2019-04-07","arxiv_id":"1904.03551","repositories_listed":0,"syntology":null},{"url":"/paper/token-level-ensemble-distillation-for","slug":"token-level-ensemble-distillation-for","title":"Token-Level Ensemble Distillation for Grapheme-to-Phoneme Conversion","date":"2019-04-06","arxiv_id":"1904.03446","repositories_listed":0,"syntology":null},{"url":null,"slug":"m2kd-multi-model-and-multi-level-knowledge","title":"M2KD: Multi-model and Multi-level Knowledge Distillation for Incremental Learning","date":"2019-04-03","arxiv_id":"1904.01769","repositories_listed":0,"syntology":null},{"url":null,"slug":"making-neural-machine-reading-comprehension","title":"Making Neural Machine Reading Comprehension Faster","date":"2019-03-29","arxiv_id":"1904.00796","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-route-choice-models-by","title":"Improving Route Choice Models by Incorporating Contextual Factors via Knowledge Distillation","date":"2019-03-27","arxiv_id":"1903.11253","repositories_listed":0,"syntology":null},{"url":null,"slug":"rectified-decision-trees-towards","title":"Rectified Decision Trees: Towards Interpretability, Compression and Empirical Soundness","date":"2019-03-14","arxiv_id":"1903.05965","repositories_listed":0,"syntology":null},{"url":null,"slug":"refine-and-distill-exploiting-cycle","title":"Refine and Distill: Exploiting Cycle-Inconsistency and Knowledge Distillation for Unsupervised Monocular Depth Estimation","date":"2019-03-11","arxiv_id":"1903.04202","repositories_listed":0,"syntology":null},{"url":null,"slug":"tkd-temporal-knowledge-distillation-for","title":"TKD: Temporal Knowledge Distillation for Active Perception","date":"2019-03-04","arxiv_id":"1903.01522","repositories_listed":0,"syntology":null},{"url":null,"slug":"micik-mining-cross-layer-inherent-similarity","title":"MICIK: MIning Cross-Layer Inherent Similarity Knowledge for Deep Model Compression","date":"2019-02-03","arxiv_id":"1902.00918","repositories_listed":0,"syntology":null},{"url":null,"slug":"compressing-gans-using-knowledge-distillation","title":"Compressing GANs using Knowledge Distillation","date":"2019-02-01","arxiv_id":"1902.00159","repositories_listed":0,"syntology":null},{"url":null,"slug":"progressive-label-distillation-learning-input","title":"Progressive Label Distillation: Learning Input-Efficient Deep Neural Networks","date":"2019-01-26","arxiv_id":"1901.09135","repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-learning-of-neural-networks-to","title":"Unsupervised Learning of Neural Networks to Explain Neural Networks (extended abstract)","date":"2019-01-21","arxiv_id":"1901.07538","repositories_listed":0,"syntology":null},{"url":null,"slug":"stealing-neural-networks-via-timing-side","title":"Stealing Neural Networks via Timing Side Channels","date":"2018-12-31","arxiv_id":"1812.11720","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-the-interpretability-of-deep-neural","title":"Improving the Interpretability of Deep Neural Networks with Knowledge Distillation","date":"2018-12-28","arxiv_id":"1812.10924","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-student-networks-via-feature","title":"Learning Student Networks via Feature Embedding","date":"2018-12-17","arxiv_id":"1812.06597","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatial-knowledge-distillation-to-aid-visual","title":"Spatial Knowledge Distillation to aid Visual Reasoning","date":"2018-12-10","arxiv_id":"1812.03631","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-speedaccuracy-trade-off-for-person","title":"Optimizing speed/accuracy trade-off for person re-identification via knowledge distillation","date":"2018-12-07","arxiv_id":"1812.02937","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-large-scale-knowledge","title":"Accelerating Large Scale Knowledge Distillation via Dynamic Importance Sampling","date":"2018-12-03","arxiv_id":"1812.00914","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-with-feature-maps-for","title":"Knowledge Distillation with Feature Maps for Image Classification","date":"2018-12-03","arxiv_id":"1812.00660","repositories_listed":0,"syntology":null},{"url":null,"slug":"kdgan-knowledge-distillation-with-generative","title":"KDGAN: Knowledge Distillation with Generative Adversarial Networks","date":"2018-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-specialize-with-knowledge","title":"Learning to Specialize with Knowledge Distillation for Visual Question Answering","date":"2018-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-compressing-u-net-using-knowledge","title":"On Compressing U-net Using Knowledge Distillation","date":"2018-12-01","arxiv_id":"1812.00249","repositories_listed":0,"syntology":null},{"url":null,"slug":"expandnets-exploiting-linear-redundancy-to","title":"ExpandNets: Linear Over-parameterization to Train Compact Convolutional Networks","date":"2018-11-26","arxiv_id":"1811.10495","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-resolution-face-recognition-in-the-wild-1","title":"Low-resolution Face Recognition in the Wild via Selective Knowledge Distillation","date":"2018-11-25","arxiv_id":"1811.09998","repositories_listed":0,"syntology":null},{"url":null,"slug":"structured-pruning-of-neural-networks-with","title":"Structured Pruning of Neural Networks with Budget-Aware Regularization","date":"2018-11-23","arxiv_id":"1811.09332","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-adaptive-pruning-for-efficient","title":"Graph-Adaptive Pruning for Efficient Inference of Convolutional Neural Networks","date":"2018-11-21","arxiv_id":"1811.08589","repositories_listed":0,"syntology":null},{"url":null,"slug":"factorized-distillation-training-holistic","title":"Factorized Distillation: Training Holistic Person Re-identification Model by Distilling an Ensemble of Partial ReID Models","date":"2018-11-20","arxiv_id":"1811.08073","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-referenced-deep-learning","title":"Self-Referenced Deep Learning","date":"2018-11-19","arxiv_id":"1811.07598","repositories_listed":0,"syntology":null},{"url":null,"slug":"private-model-compression-via-knowledge","title":"Private Model Compression via Knowledge Distillation","date":"2018-11-13","arxiv_id":"1811.05072","repositories_listed":0,"syntology":null},{"url":null,"slug":"sequence-level-knowledge-distillation-for","title":"Sequence-Level Knowledge Distillation for Model Compression of Attention-based Sequence-to-Sequence Speech Recognition","date":"2018-11-12","arxiv_id":"1811.04531","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-closer-look-at-deep-learning-heuristics-1","title":"A Closer Look at Deep Learning Heuristics: Learning rate restarts, Warmup and Distillation","date":"2018-10-29","arxiv_id":"1810.13243","repositories_listed":0,"syntology":null},{"url":null,"slug":"block-wise-intermediate-representation","title":"Block-wise Intermediate Representation Training for Model Compression","date":"2018-10-20","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"ktan-knowledge-transfer-adversarial-network","title":"KTAN: Knowledge Transfer Adversarial Network","date":"2018-10-18","arxiv_id":"1810.08126","repositories_listed":0,"syntology":null},{"url":null,"slug":"lit-block-wise-intermediate-representation","title":"LIT: Block-wise Intermediate Representation Training for Model Compression","date":"2018-10-02","arxiv_id":"1810.01937","repositories_listed":0,"syntology":null},{"url":null,"slug":"analyzing-knowledge-distillation-in-neural","title":"Analyzing Knowledge Distillation in Neural Machine Translation","date":"2018-10-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"5099e4b42b802adeea4ea9f4a2060e5e6e0062b54b13e983dd93413b3ec5a17e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}