{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/knowledge-distillation/papers/30","list_of":"/method/knowledge-distillation","method":"Knowledge Distillation","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":30,"pages_in_order":31,"rows_per_page":100,"rows":[2901,3000],"of":3071,"counts":{"archive_papers_tagged":3071,"with_a_code_link":1258,"where_syntology_ran_a_sample":320,"not_listed_spam_title":0,"listed":3071,"listed_where_code_ran":320,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":276,"every_run_a_failure_of_syntologys_instrument":44,"listed_with_a_run_with_no_instrument_failure":276,"listed_every_run_a_failure_of_syntologys_instrument":44,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/knowledge-distillation","prev":"/method/knowledge-distillation/papers/29","next":"/method/knowledge-distillation/papers/31","papers":[{"paper":null,"slug":"acquiring-knowledge-from-pre-trained-model-to","title":"Acquiring Knowledge from Pre-trained Model to Neural Machine Translation","date":"2019-12-04","arxiv_id":"1912.01774","n_code_links":0,"syntology":null},{"paper":"/paper/quest-quantized-embedding-space-for","slug":"quest-quantized-embedding-space-for","title":"QUEST: Quantized embedding space for transferring knowledge","date":"2019-12-03","arxiv_id":"1912.01540","n_code_links":1,"syntology":null},{"paper":null,"slug":"efficient-convolutional-neural-networks-for-2","title":"Efficient Convolutional Neural Networks for Depth-Based Multi-Person Pose Estimation","date":"2019-12-02","arxiv_id":"1912.00711","n_code_links":0,"syntology":null},{"paper":"/paper/online-knowledge-distillation-with-diverse","slug":"online-knowledge-distillation-with-diverse","title":"Online Knowledge Distillation with Diverse Peers","date":"2019-12-01","arxiv_id":"1912.00350","n_code_links":2,"syntology":null},{"paper":"/paper/random-path-selection-for-continual-learning","slug":"random-path-selection-for-continual-learning","title":"Random Path Selection for Continual Learning","date":"2019-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-oracle-knowledge-distillation-with","title":"Towards Oracle Knowledge Distillation with Neural Architecture Search","date":"2019-11-29","arxiv_id":"1911.13019","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-driven-compression-of-convolutional","title":"Data-Driven Compression of Convolutional Neural Networks","date":"2019-11-28","arxiv_id":"1911.12740","n_code_links":0,"syntology":null},{"paper":null,"slug":"qkd-quantization-aware-knowledge-distillation","title":"QKD: Quantization-aware Knowledge Distillation","date":"2019-11-28","arxiv_id":"1911.12491","n_code_links":0,"syntology":null},{"paper":"/paper/go-from-the-general-to-the-particular-multi","slug":"go-from-the-general-to-the-particular-multi","title":"Go From the General to the Particular: Multi-Domain Translation with Domain Transformation Networks","date":"2019-11-22","arxiv_id":"1911.09912","n_code_links":2,"syntology":null},{"paper":"/paper/few-shot-network-compression-via-cross","slug":"few-shot-network-compression-via-cross","title":"Few Shot Network Compression via Cross Distillation","date":"2019-11-21","arxiv_id":"1911.09450","n_code_links":1,"syntology":null},{"paper":null,"slug":"search-to-distill-pearls-are-everywhere-but","title":"Search to Distill: Pearls are Everywhere but not the Eyes","date":"2019-11-20","arxiv_id":"1911.09074","n_code_links":0,"syntology":null},{"paper":"/paper/neural-network-pruning-with-residual","slug":"neural-network-pruning-with-residual","title":"Neural Network Pruning with Residual-Connections and Limited-Data","date":"2019-11-19","arxiv_id":"1911.08114","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":0,"n_instrument":4,"unverified":3,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/maintaining-discrimination-and-fairness-in","slug":"maintaining-discrimination-and-fairness-in","title":"Maintaining Discrimination and Fairness in Class Incremental Learning","date":"2019-11-16","arxiv_id":"1911.07053","n_code_links":2,"syntology":null},{"paper":"/paper/stagewise-knowledge-distillation","slug":"stagewise-knowledge-distillation","title":"Data Efficient Stagewise Knowledge Distillation","date":"2019-11-15","arxiv_id":"1911.06786","n_code_links":1,"syntology":null},{"paper":"/paper/knowledge-representing-efficient-sparse","slug":"knowledge-representing-efficient-sparse","title":"Knowledge Representing: Efficient, Sparse Representation of Prior Knowledge for Knowledge Distillation","date":"2019-11-13","arxiv_id":"1911.05329","n_code_links":0,"syntology":null},{"paper":"/paper/learning-from-a-teacher-using-unlabeled-data","slug":"learning-from-a-teacher-using-unlabeled-data","title":"Learning from a Teacher using Unlabeled Data","date":"2019-11-13","arxiv_id":"1911.05275","n_code_links":1,"syntology":null},{"paper":null,"slug":"graph-representation-learning-via-multi-task","title":"Graph Representation Learning via Multi-task Knowledge Distillation","date":"2019-11-11","arxiv_id":"1911.05700","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-distillation-in-document-retrieval","title":"Knowledge Distillation in Document Retrieval","date":"2019-11-11","arxiv_id":"1911.11065","n_code_links":0,"syntology":null},{"paper":null,"slug":"attentive-student-meets-multi-task-teacher","title":"MKD: a Multi-Task Knowledge Distillation Approach for Pretrained Language Models","date":"2019-11-09","arxiv_id":"1911.03588","n_code_links":0,"syntology":null},{"paper":"/paper/deep-geometric-knowledge-distillation-with","slug":"deep-geometric-knowledge-distillation-with","title":"Deep geometric knowledge distillation with graphs","date":"2019-11-08","arxiv_id":"1911.03080","n_code_links":1,"syntology":null},{"paper":null,"slug":"knowledge-distillation-for-incremental","title":"Knowledge Distillation for Incremental Learning in Semantic Segmentation","date":"2019-11-08","arxiv_id":"1911.03462","n_code_links":0,"syntology":null},{"paper":null,"slug":"understanding-knowledge-distillation-in-non","title":"Understanding Knowledge Distillation in Non-autoregressive Machine Translation","date":"2019-11-07","arxiv_id":"1911.02727","n_code_links":0,"syntology":null},{"paper":"/paper/data-diversification-an-elegant-strategy-for","slug":"data-diversification-an-elegant-strategy-for","title":"Data Diversification: A Simple Strategy For Neural Machine Translation","date":"2019-11-05","arxiv_id":"1911.01986","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":0,"n_instrument":3,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["nxphi47/data_diversification"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"distilling-pixel-wise-feature-similarities","title":"Distilling Pixel-Wise Feature Similarities for Semantic Segmentation","date":"2019-10-31","arxiv_id":"1910.14226","n_code_links":0,"syntology":null},{"paper":"/paper/a-simple-but-effective-bert-model-for-dialog","slug":"a-simple-but-effective-bert-model-for-dialog","title":"A Simple but Effective BERT Model for Dialog State Tracking on Resource-Limited Systems","date":"2019-10-28","arxiv_id":"1910.12995","n_code_links":0,"syntology":null},{"paper":"/paper/mod-a-deep-mixture-model-with-online","slug":"mod-a-deep-mixture-model-with-online","title":"MOD: A Deep Mixture Model with Online Knowledge Distillation for Large Scale Video Temporal Concept Localization","date":"2019-10-27","arxiv_id":"1910.12295","n_code_links":1,"syntology":null},{"paper":null,"slug":"variational-student-learning-compact-and","title":"Variational Student: Learning Compact and Sparser Networks in Knowledge Distillation Framework","date":"2019-10-26","arxiv_id":"1910.12061","n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarial-feature-alignment-avoid","title":"Adversarial Feature Alignment: Avoid Catastrophic Forgetting in Incremental Task Lifelong Learning","date":"2019-10-24","arxiv_id":"1910.10986","n_code_links":0,"syntology":null},{"paper":"/paper/contrastive-representation-distillation-1","slug":"contrastive-representation-distillation-1","title":"Contrastive Representation Distillation","date":"2019-10-23","arxiv_id":"1910.10699","n_code_links":4,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["HobbitLong/RepDistiller"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":null,"slug":"model-compression-with-two-stage-multi","title":"Model Compression with Two-stage Multi-teacher Knowledge Distillation for Web Question Answering System","date":"2019-10-18","arxiv_id":"1910.08381","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-generalized-and-robust-method-towards","title":"A Generalized and Robust Method Towards Practical Gaze Estimation on Smart Phone","date":"2019-10-16","arxiv_id":"1910.07331","n_code_links":0,"syntology":null},{"paper":null,"slug":"improving-generalization-and-robustness-with","title":"Noise as a Resource for Learning in Knowledge Distillation","date":"2019-10-11","arxiv_id":"1910.05057","n_code_links":0,"syntology":null},{"paper":"/paper/vargfacenet-an-efficient-variable-group","slug":"vargfacenet-an-efficient-variable-group","title":"VarGFaceNet: An Efficient Variable Group Convolutional Neural Network for Lightweight Face Recognition","date":"2019-10-11","arxiv_id":"1910.04985","n_code_links":3,"syntology":null},{"paper":null,"slug":"cross-modal-knowledge-distillation-for-action","title":"Cross-modal knowledge distillation for action recognition","date":"2019-10-10","arxiv_id":"1910.04641","n_code_links":0,"syntology":null},{"paper":"/paper/fedmd-heterogenous-federated-learning-via","slug":"fedmd-heterogenous-federated-learning-via","title":"FedMD: Heterogenous Federated Learning via Model Distillation","date":"2019-10-08","arxiv_id":"1910.03581","n_code_links":7,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":null,"slug":"knowledge-distillation-from-internal","title":"Knowledge Distillation from Internal Representations","date":"2019-10-08","arxiv_id":"1910.03723","n_code_links":0,"syntology":null},{"paper":null,"slug":"distilling-transformers-into-simple-neural","title":"Distilling BERT into Simple Neural Networks with Unlabeled Transfer Data","date":"2019-10-04","arxiv_id":"1910.01769","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-efficacy-of-knowledge-distillation","title":"On the Efficacy of Knowledge Distillation","date":"2019-10-03","arxiv_id":"1910.01348","n_code_links":0,"syntology":null},{"paper":null,"slug":"antman-sparse-low-rank-compression-to-1","title":"AntMan: Sparse Low-Rank Compression to Accelerate RNN inference","date":"2019-10-02","arxiv_id":"1910.01740","n_code_links":0,"syntology":null},{"paper":"/paper/distilbert-a-distilled-version-of-bert","slug":"distilbert-a-distilled-version-of-bert","title":"DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter","date":"2019-10-02","arxiv_id":"1910.01108","n_code_links":37,"syntology":{"ran":21,"of":27,"n_ran_checked":13,"n_instrument":8,"unverified":6,"pointer_only":2,"phrase":"21 ran (of which 5 constructed an object rather than computing a result; 13 with no instrument failure: 3 honoured, 1 violated, 9 with no contract checked; 8 where Syntology's instrument failed) · 6 unverified","official":{"repos":["huggingface/swift-coreml-transformers","huggingface/transformers"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"distilled-embedding-non-linear-embedding","title":"Improving Word Embedding Factorization for Compression Using Distilled Nonlinear Neural Decomposition","date":"2019-10-02","arxiv_id":"1910.06720","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-bayesian-optimization-framework-for-neural","title":"A Bayesian Optimization Framework for Neural Network Compression","date":"2019-10-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/distilled-split-deep-neural-networks-for-edge","slug":"distilled-split-deep-neural-networks-for-edge","title":"Distilled Split Deep Neural Networks for Edge-Assisted Real-Time Systems","date":"2019-10-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":"/paper/training-convolutional-neural-networks-with-2","slug":"training-convolutional-neural-networks-with-2","title":"Training convolutional neural networks with cheap convolutions and online distillation","date":"2019-09-28","arxiv_id":"1909.13063","n_code_links":1,"syntology":null},{"paper":"/paper/compact-trilinear-interaction-for-visual","slug":"compact-trilinear-interaction-for-visual","title":"Compact Trilinear Interaction for Visual Question Answering","date":"2019-09-26","arxiv_id":"1909.11874","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["aioz-ai/ICCV19_VQA-CTI"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"collaborative-inter-agent-knowledge","title":"Collaborative Inter-agent Knowledge Distillation for Reinforcement Learning","date":"2019-09-25","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"distilled-embedding-non-linear-embedding-1","title":"Distilled embedding: non-linear embedding factorization using knowledge distillation","date":"2019-09-25","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"extreme-language-model-compression-with-1","title":"Extremely Small BERT Models from Mixed-Vocabulary Training","date":"2019-09-25","arxiv_id":"1909.11687","n_code_links":0,"syntology":null},{"paper":null,"slug":"proactive-sequence-generator-via-knowledge","title":"Proactive Sequence Generator via Knowledge Acquisition","date":"2019-09-25","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/revisit-knowledge-distillation-a-teacher-free","slug":"revisit-knowledge-distillation-a-teacher-free","title":"Revisiting Knowledge Distillation via Label Smoothing Regularization","date":"2019-09-25","arxiv_id":"1909.11723","n_code_links":2,"syntology":{"ran":11,"of":13,"n_ran_checked":11,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["yuanli2333/Teacher-free-Knowledge-Distillation"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"self-knowledge-distillation-adversarial","title":"SELF-KNOWLEDGE DISTILLATION ADVERSARIAL ATTACK","date":"2019-09-25","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"xd-cross-lingual-knowledge-distillation-for","title":"XD: Cross-lingual Knowledge Distillation for Polyglot Sentence Embeddings","date":"2019-09-25","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"feed-feature-level-ensemble-for-knowledge","title":"FEED: Feature-level Ensemble for Knowledge Distillation","date":"2019-09-24","arxiv_id":"1909.10754","n_code_links":0,"syntology":null},{"paper":null,"slug":"technical-report-on-conversational-question","title":"Technical report on Conversational Question Answering","date":"2019-09-24","arxiv_id":"1909.10772","n_code_links":0,"syntology":null},{"paper":"/paper/190910351","slug":"190910351","title":"TinyBERT: Distilling BERT for Natural Language Understanding","date":"2019-09-23","arxiv_id":"1909.10351","n_code_links":10,"syntology":{"ran":0,"of":4,"n_ran_checked":0,"n_instrument":0,"unverified":4,"pointer_only":4,"phrase":"0 ran · 4 unverified","official":{"repos":["huawei-noah/Pretrained-Language-Model"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":[]}}},{"paper":"/paper/190909757","slug":"190909757","title":"Positive-Unlabeled Compression on the Cloud","date":"2019-09-21","arxiv_id":"1909.09757","n_code_links":2,"syntology":null},{"paper":null,"slug":"learning-lightweight-pedestrian-detector-with","title":"Learning Lightweight Pedestrian Detector with Hierarchical Knowledge Distillation","date":"2019-09-20","arxiv_id":"1909.09325","n_code_links":0,"syntology":null},{"paper":"/paper/ensemble-knowledge-distillation-for-learning","slug":"ensemble-knowledge-distillation-for-learning","title":"Ensemble Knowledge Distillation for Learning Improved and Efficient Networks","date":"2019-09-17","arxiv_id":"1909.08097","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/knowledge-distillation-for-end-to-endperson","slug":"knowledge-distillation-for-end-to-endperson","title":"Knowledge Distillation for End-to-End Person Search","date":"2019-09-03","arxiv_id":"1909.01058","n_code_links":1,"syntology":null},{"paper":null,"slug":"online-sensor-hallucination-via-knowledge","title":"Online Sensor Hallucination via Knowledge Distillation for Multimodal Image Classification","date":"2019-08-28","arxiv_id":"1908.10559","n_code_links":0,"syntology":null},{"paper":"/paper/patient-knowledge-distillation-for-bert-model","slug":"patient-knowledge-distillation-for-bert-model","title":"Patient Knowledge Distillation for BERT Model Compression","date":"2019-08-25","arxiv_id":"1908.09355","n_code_links":5,"syntology":{"ran":19,"of":27,"n_ran_checked":13,"n_instrument":6,"unverified":8,"pointer_only":27,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 2 honoured, 1 violated, 10 with no contract checked; 6 where Syntology's instrument failed) · 8 unverified","official":{"repos":["intersun/PKD-for-BERT-Model-Compression"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/improved-techniques-for-training-adaptive","slug":"improved-techniques-for-training-adaptive","title":"Improved Techniques for Training Adaptive Deep Networks","date":"2019-08-17","arxiv_id":"1908.06294","n_code_links":2,"syntology":null},{"paper":null,"slug":"knowledge-distillation-for-semi-supervised","title":"Knowledge distillation for semi-supervised domain adaptation","date":"2019-08-16","arxiv_id":"1908.07355","n_code_links":0,"syntology":null},{"paper":"/paper/scalable-attentive-sentence-pair-modeling-via","slug":"scalable-attentive-sentence-pair-modeling-via","title":"Scalable Attentive Sentence-Pair Modeling via Distilled Sentence Embedding","date":"2019-08-14","arxiv_id":"1908.05161","n_code_links":1,"syntology":null},{"paper":null,"slug":"effective-training-of-convolutional-neural","title":"Effective Training of Convolutional Neural Networks with Low-bitwidth Weights and Activations","date":"2019-08-10","arxiv_id":"1908.04680","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-isomorphism-between-neural-networks","title":"Knowledge Consistency between Neural Networks and Beyond","date":"2019-08-05","arxiv_id":"1908.01581","n_code_links":0,"syntology":null},{"paper":"/paper/learning-lightweight-lane-detection-cnns-by","slug":"learning-lightweight-lane-detection-cnns-by","title":"Learning Lightweight Lane Detection CNNs by Self Attention Distillation","date":"2019-08-02","arxiv_id":"1908.00821","n_code_links":2,"syntology":{"ran":7,"of":8,"n_ran_checked":5,"n_instrument":2,"unverified":1,"pointer_only":1,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cardwing/Codes-for-Lane-Detection"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"self-knowledge-distillation-in-natural","title":"Self-Knowledge Distillation in Natural Language Processing","date":"2019-08-02","arxiv_id":"1908.01851","n_code_links":0,"syntology":null},{"paper":null,"slug":"gtcom-neural-machine-translation-systems-for","title":"GTCOM Neural Machine Translation Systems for WMT19","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"panlp-at-mediqa-2019-pre-trained-language","title":"PANLP at MEDIQA 2019: Pre-trained Language Models, Transfer Learning and Knowledge Distillation","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"the-niutrans-machine-translation-systems-for","title":"The NiuTrans Machine Translation Systems for WMT19","date":"2019-08-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"distill-to-label-weakly-supervised-instance","title":"Distill-to-Label: Weakly Supervised Instance Labeling Using Knowledge Distillation","date":"2019-07-26","arxiv_id":"1907.12926","n_code_links":0,"syntology":null},{"paper":null,"slug":"teacher-students-knowledge-distillation-for","title":"Distilled Siamese Networks for Visual Tracking","date":"2019-07-24","arxiv_id":"1907.10586","n_code_links":0,"syntology":null},{"paper":"/paper/highlight-every-step-knowledge-distillation","slug":"highlight-every-step-knowledge-distillation","title":"Highlight Every Step: Knowledge Distillation via Collaborative Teaching","date":"2019-07-23","arxiv_id":"1907.09643","n_code_links":1,"syntology":null},{"paper":null,"slug":"lifelong-gan-continual-learning-for","title":"Lifelong GAN: Continual Learning for Conditional Image Generation","date":"2019-07-23","arxiv_id":"1907.10107","n_code_links":0,"syntology":null},{"paper":"/paper/real-time-correlation-tracking-via-joint","slug":"real-time-correlation-tracking-via-joint","title":"Real-Time Correlation Tracking via Joint Model Compression and Transfer","date":"2019-07-23","arxiv_id":"1907.09831","n_code_links":1,"syntology":null},{"paper":"/paper/similarity-preserving-knowledge-distillation","slug":"similarity-preserving-knowledge-distillation","title":"Similarity-Preserving Knowledge Distillation","date":"2019-07-23","arxiv_id":"1907.09682","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/light-multi-segment-activation-for-model","slug":"light-multi-segment-activation-for-model","title":"Light Multi-segment Activation for Model Compression","date":"2019-07-16","arxiv_id":"1907.06870","n_code_links":2,"syntology":null},{"paper":null,"slug":"learn-spelling-from-teachers-transferring","title":"Learn Spelling from Teachers: Transferring Knowledge from Language Models to Sequence-to-Sequence Speech Recognition","date":"2019-07-13","arxiv_id":"1907.06017","n_code_links":0,"syntology":null},{"paper":"/paper/bam-born-again-multi-task-networks-for","slug":"bam-born-again-multi-task-networks-for","title":"BAM! Born-Again Multi-Task Networks for Natural Language Understanding","date":"2019-07-10","arxiv_id":"1907.04829","n_code_links":1,"syntology":null},{"paper":null,"slug":"compression-of-acoustic-event-detection-1","title":"Compression of Acoustic Event Detection Models With Quantized Distillation","date":"2019-07-01","arxiv_id":"1907.00873","n_code_links":0,"syntology":null},{"paper":"/paper/approximating-interactive-human-evaluation","slug":"approximating-interactive-human-evaluation","title":"Approximating Interactive Human Evaluation with Self-Play for Open-Domain Dialog Systems","date":"2019-06-21","arxiv_id":"1906.09308","n_code_links":2,"syntology":null},{"paper":null,"slug":"gan-knowledge-distillation-for-one-stage","title":"GAN-Knowledge Distillation for one-stage Object Detection","date":"2019-06-20","arxiv_id":"1906.08467","n_code_links":0,"syntology":null},{"paper":null,"slug":"reconciling-utility-and-membership-privacy","title":"Membership Privacy for Machine Learning Models Through Knowledge Transfer","date":"2019-06-15","arxiv_id":"1906.06589","n_code_links":0,"syntology":null},{"paper":null,"slug":"divide-and-conquer-leveraging-intermediate","title":"Divide and Conquer: Leveraging Intermediate Feature Representations for Quantized Training of Neural Networks","date":"2019-06-14","arxiv_id":"1906.06033","n_code_links":0,"syntology":null},{"paper":null,"slug":"scalable-syntax-aware-language-models-using","title":"Scalable Syntax-Aware Language Models Using Knowledge Distillation","date":"2019-06-14","arxiv_id":"1906.06438","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-evaluation-time-uncertainty","title":"Efficient Evaluation-Time Uncertainty Estimation by Improved Distillation","date":"2019-06-12","arxiv_id":"1906.05419","n_code_links":0,"syntology":null},{"paper":"/paper/distilling-object-detectors-with-fine-grained-1","slug":"distilling-object-detectors-with-fine-grained-1","title":"Distilling Object Detectors with Fine-grained Feature Imitation","date":"2019-06-09","arxiv_id":"1906.03609","n_code_links":3,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["twangnh/Distilling-Object-Detectors"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/when-does-label-smoothing-help","slug":"when-does-label-smoothing-help","title":"When Does Label Smoothing Help?","date":"2019-06-06","arxiv_id":"1906.02629","n_code_links":4,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":null,"slug":"private-deep-learning-with-teacher-ensembles","title":"Private Deep Learning with Teacher Ensembles","date":"2019-06-05","arxiv_id":"1906.02303","n_code_links":0,"syntology":null},{"paper":null,"slug":"190600619","title":"Deep Face Recognition Model Compression via Knowledge Transfer and Distillation","date":"2019-06-03","arxiv_id":"1906.00619","n_code_links":0,"syntology":null},{"paper":"/paper/random-path-selection-for-incremental","slug":"random-path-selection-for-incremental","title":"An Adaptive Random Path Selection Approach for Incremental Learning","date":"2019-06-03","arxiv_id":"1906.01120","n_code_links":1,"syntology":null},{"paper":"/paper/knowledge-distillation-via-instance","slug":"knowledge-distillation-via-instance","title":"Knowledge Distillation via Instance Relationship Graph","date":"2019-06-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"on-knowledge-distillation-from-complex","title":"On Knowledge distillation from complex networks for response prediction","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"online-distilling-from-checkpoints-for-neural","title":"Online Distilling from Checkpoints for Neural Machine Translation","date":"2019-06-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/structured-knowledge-distillation-for-1","slug":"structured-knowledge-distillation-for-1","title":"Structured Knowledge Distillation for Semantic Segmentation","date":"2019-06-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/scan-a-scalable-neural-networks-framework","slug":"scan-a-scalable-neural-networks-framework","title":"SCAN: A Scalable Neural Networks Framework Towards Compact and Efficient Models","date":"2019-05-27","arxiv_id":"1906.03951","n_code_links":1,"syntology":null},{"paper":null,"slug":"cross-resolution-face-recognition-via-prior","title":"Cross-Resolution Face Recognition via Prior-Aided Face Hallucination and Residual Knowledge Distillation","date":"2019-05-26","arxiv_id":"1905.10777","n_code_links":0,"syntology":null},{"paper":"/paper/be-your-own-teacher-improve-the-performance","slug":"be-your-own-teacher-improve-the-performance","title":"Be Your Own Teacher: Improve the Performance of Convolutional Neural Networks via Self Distillation","date":"2019-05-17","arxiv_id":"1905.08094","n_code_links":1,"syntology":null},{"paper":null,"slug":"creating-lightweight-object-detectors-with","title":"Creating Lightweight Object Detectors with Model Compression for Deployment on Edge Devices","date":"2019-05-06","arxiv_id":"1905.01787","n_code_links":0,"syntology":null}],"record_sha256":"7743b94fba57159fae8042d26b66522ed9ba2997b1668df277f05e9835ef518e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}