{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/knowledge-distillation/papers/27","list_of":"/method/knowledge-distillation","method":"Knowledge Distillation","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":27,"pages_in_order":31,"rows_per_page":100,"rows":[2601,2700],"of":3071,"counts":{"archive_papers_tagged":3071,"with_a_code_link":1258,"where_syntology_ran_a_sample":320,"not_listed_spam_title":0,"listed":3071,"listed_where_code_ran":320,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":276,"every_run_a_failure_of_syntologys_instrument":44,"listed_with_a_run_with_no_instrument_failure":276,"listed_every_run_a_failure_of_syntologys_instrument":44,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/knowledge-distillation","prev":"/method/knowledge-distillation/papers/26","next":"/method/knowledge-distillation/papers/28","papers":[{"paper":null,"slug":"deep-epidemiological-modeling-by-black-box","title":"Deep Epidemiological Modeling by Black-box Knowledge Distillation: An Accurate Deep Learning Model for COVID-19","date":"2021-01-20","arxiv_id":"2101.10280","n_code_links":0,"syntology":null},{"paper":"/paper/learning-to-augment-for-data-scarce-domain","slug":"learning-to-augment-for-data-scarce-domain","title":"Learning to Augment for Data-Scarce Domain BERT Knowledge Distillation","date":"2021-01-20","arxiv_id":"2101.08106","n_code_links":0,"syntology":{"ran":2,"of":15,"n_ran_checked":2,"n_instrument":0,"unverified":13,"pointer_only":15,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 13 unverified","official":null}},{"paper":null,"slug":"knowledge-distillation-methods-for-efficient","title":"Knowledge Distillation Methods for Efficient Unsupervised Adaptation Across Multiple Domains","date":"2021-01-18","arxiv_id":"2101.07308","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-impressions-mining-deep-models-to","title":"Mining Data Impressions from Deep Models as Substitute for the Unavailable Training Data","date":"2021-01-15","arxiv_id":"2101.06069","n_code_links":0,"syntology":null},{"paper":null,"slug":"kdlsq-bert-a-quantized-bert-combining","title":"KDLSQ-BERT: A Quantized Bert Combining Knowledge Distillation with Learned Step Size Quantization","date":"2021-01-15","arxiv_id":"2101.05938","n_code_links":0,"syntology":null},{"paper":null,"slug":"interpretable-discovery-of-new-semiconductors","title":"Interpretable discovery of new semiconductors with machine learning","date":"2021-01-12","arxiv_id":"2101.04383","n_code_links":0,"syntology":null},{"paper":null,"slug":"resolution-based-distillation-for-efficient","title":"Resolution-Based Distillation for Efficient Histology Image Classification","date":"2021-01-11","arxiv_id":"2101.04170","n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarially-robust-and-explainable-model","title":"Adversarially Robust and Explainable Model Compression with On-Device Personalization for Text Classification","date":"2021-01-10","arxiv_id":"2101.05624","n_code_links":0,"syntology":null},{"paper":"/paper/knowledge-distillation-in-iterative","slug":"knowledge-distillation-in-iterative","title":"Knowledge Distillation in Iterative Generative Models for Improved Sampling Speed","date":"2021-01-07","arxiv_id":"2101.02388","n_code_links":2,"syntology":null},{"paper":null,"slug":"modality-specific-distillation","title":"MSD: Saliency-aware Knowledge Distillation for Multimodal Understanding","date":"2021-01-06","arxiv_id":"2101.01881","n_code_links":0,"syntology":null},{"paper":null,"slug":"active-learning-for-lane-detection-a","title":"Active Learning for Lane Detection: A Knowledge Distillation Approach","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"contextual-knowledge-distillation-for","title":"Contextual Knowledge Distillation for Transformer Compression","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"don-t-be-picky-all-students-in-the-right","title":"Don't be picky, all students in the right family can learn from good teachers","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"explicit-connection-distillation","title":"Explicit Connection Distillation","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/exploring-inter-channel-correlation-for","slug":"exploring-inter-channel-correlation-for","title":"Exploring Inter-Channel Correlation for Diversity-Preserved Knowledge Distillation","date":"2021-01-01","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":"/paper/improve-object-detection-with-feature-based","slug":"improve-object-detection-with-feature-based","title":"Improve Object Detection with Feature-based Knowledge Distillation: Towards Accurate and Efficient Detectors","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"knowledge-distillation-based-ensemble","title":"Knowledge Distillation based Ensemble Learning for Neural Machine Translation","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-from-deep-model-via-exploring-local","title":"Learning from deep model via exploring local targets","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"long-live-the-lottery-the-existence-of","title":"Long Live the Lottery: The Existence of Winning Tickets in Lifelong Learning","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"robust-overfitting-may-be-mitigated-by","title":"Robust Overfitting may be mitigated by properly learned smoothening","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/self-mutual-distillation-learning-for","slug":"self-mutual-distillation-learning-for","title":"Self-Mutual Distillation Learning for Continuous Sign Language Recognition","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/stem-an-approach-to-multi-source-domain","slug":"stem-an-approach-to-multi-source-domain","title":"STEM: An Approach to Multi-Source Domain Adaptation With Guarantees","date":"2021-01-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"student-customized-knowledge-distillation","title":"Student Customized Knowledge Distillation: Bridging the Gap Between Student and Teacher","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"unpaired-learning-for-deep-image-deraining","title":"Unpaired Learning for Deep Image Deraining With Rain Direction Regularizer","date":"2021-01-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-monolingual-data-for-neural-machine","title":"Fully Synthetic Data Improves Neural Machine Translation with Knowledge Distillation","date":"2020-12-31","arxiv_id":"2012.15455","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-zero-shot-knowledge-distillation-for","title":"Towards Zero-Shot Knowledge Distillation for Natural Language Processing","date":"2020-12-31","arxiv_id":"2012.15495","n_code_links":0,"syntology":null},{"paper":"/paper/unified-mandarin-tts-front-end-based-on","slug":"unified-mandarin-tts-front-end-based-on","title":"Unified Mandarin TTS Front-end Based on Distilled BERT Model","date":"2020-12-31","arxiv_id":"2012.15404","n_code_links":1,"syntology":null},{"paper":null,"slug":"knowledge-distillation-with-adaptive","title":"Knowledge Distillation with Adaptive Asymmetric Label Sharpening for Semi-supervised Fracture Detection in Chest X-rays","date":"2020-12-30","arxiv_id":"2012.15359","n_code_links":0,"syntology":null},{"paper":"/paper/learning-light-weight-translation-models-from","slug":"learning-light-weight-translation-models-from","title":"Learning Light-Weight Translation Models from Deep Transformer","date":"2020-12-27","arxiv_id":"2012.13866","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":1,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":{"repos":["libeineu/GPKD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"towards-a-universal-continuous-knowledge-base","title":"Towards a Universal Continuous Knowledge Base","date":"2020-12-25","arxiv_id":"2012.13568","n_code_links":0,"syntology":null},{"paper":null,"slug":"attentionlite-towards-efficient-self","title":"AttentionLite: Towards Efficient Self-Attention Models for Vision","date":"2020-12-21","arxiv_id":"2101.05216","n_code_links":0,"syntology":null},{"paper":null,"slug":"diverse-knowledge-distillation-for-end-to-end","title":"Diverse Knowledge Distillation for End-to-End Person Search","date":"2020-12-21","arxiv_id":"2012.11187","n_code_links":0,"syntology":null},{"paper":"/paper/computation-efficient-knowledge-distillation","slug":"computation-efficient-knowledge-distillation","title":"Computation-Efficient Knowledge Distillation via Uncertainty-Aware Mixup","date":"2020-12-17","arxiv_id":"2012.09413","n_code_links":1,"syntology":null},{"paper":"/paper/invariant-teacher-and-equivariant-student-for","slug":"invariant-teacher-and-equivariant-student-for","title":"Invariant Teacher and Equivariant Student for Unsupervised 3D Human Pose Estimation","date":"2020-12-17","arxiv_id":"2012.09398","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-understanding-ensemble-knowledge","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","date":"2020-12-17","arxiv_id":"2012.09816","n_code_links":0,"syntology":null},{"paper":null,"slug":"distilling-optimal-neural-networks-rapid","title":"Distilling Optimal Neural Networks: Rapid Search in Diverse Spaces","date":"2020-12-16","arxiv_id":"2012.08859","n_code_links":0,"syntology":null},{"paper":"/paper/wasserstein-contrastive-representation","slug":"wasserstein-contrastive-representation","title":"Wasserstein Contrastive Representation Distillation","date":"2020-12-15","arxiv_id":"2012.08674","n_code_links":0,"syntology":null},{"paper":null,"slug":"lrc-bert-latent-representation-contrastive","title":"LRC-BERT: Latent-representation Contrastive Knowledge Distillation for Natural Language Understanding","date":"2020-12-14","arxiv_id":"2012.07335","n_code_links":0,"syntology":null},{"paper":null,"slug":"periocular-in-the-wild-embedding-learning","title":"Periocular Embedding Learning with Consistent Knowledge Distillation from Face","date":"2020-12-12","arxiv_id":"2012.06746","n_code_links":0,"syntology":null},{"paper":null,"slug":"reinforced-multi-teacher-selection-for","title":"Reinforced Multi-Teacher Selection for Knowledge Distillation","date":"2020-12-11","arxiv_id":"2012.06048","n_code_links":0,"syntology":null},{"paper":"/paper/progressive-network-grafting-for-few-shot","slug":"progressive-network-grafting-for-few-shot","title":"Progressive Network Grafting for Few-Shot Knowledge Distillation","date":"2020-12-09","arxiv_id":"2012.04915","n_code_links":2,"syntology":null},{"paper":"/paper/de-rrd-a-knowledge-distillation-framework-for","slug":"de-rrd-a-knowledge-distillation-framework-for","title":"DE-RRD: A Knowledge Distillation Framework for Recommender System","date":"2020-12-08","arxiv_id":"2012.04357","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["SeongKu-Kang/DE-RRD_CIKM20"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":null,"slug":"model-compression-using-optimal-transport","title":"Model Compression Using Optimal Transport","date":"2020-12-07","arxiv_id":"2012.03907","n_code_links":0,"syntology":null},{"paper":"/paper/cross-layer-distillation-with-semantic","slug":"cross-layer-distillation-with-semantic","title":"Cross-Layer Distillation with Semantic Calibration","date":"2020-12-06","arxiv_id":"2012.03236","n_code_links":2,"syntology":null},{"paper":null,"slug":"multi-head-knowledge-distillation-for-model","title":"Multi-head Knowledge Distillation for Model Compression","date":"2020-12-05","arxiv_id":"2012.02911","n_code_links":0,"syntology":null},{"paper":"/paper/parallel-blockwise-knowledge-distillation-for","slug":"parallel-blockwise-knowledge-distillation-for","title":"Parallel Blockwise Knowledge Distillation for Deep Neural Network Compression","date":"2020-12-05","arxiv_id":"2012.03096","n_code_links":1,"syntology":null},{"paper":"/paper/reciprocal-supervised-learning-improves","slug":"reciprocal-supervised-learning-improves","title":"Reciprocal Supervised Learning Improves Neural Machine Translation","date":"2020-12-05","arxiv_id":"2012.02975","n_code_links":1,"syntology":null},{"paper":"/paper/meta-kd-a-meta-knowledge-distillation","slug":"meta-kd-a-meta-knowledge-distillation","title":"Meta-KD: A Meta Knowledge Distillation Framework for Language Model Compression across Domains","date":"2020-12-02","arxiv_id":"2012.01266","n_code_links":1,"syntology":null},{"paper":"/paper/agree-to-disagree-adaptive-ensemble-knowledge","slug":"agree-to-disagree-adaptive-ensemble-knowledge","title":"Agree to Disagree: Adaptive Ensemble Knowledge Distillation in Gradient Space","date":"2020-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"classification-under-misspecification-1","title":"Classification Under Misspecification: Halfspaces, Generalized Linear Models, and Evolvability","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/knowledge-base-embedding-by-cooperative","slug":"knowledge-base-embedding-by-cooperative","title":"Knowledge Base Embedding By Cooperative Knowledge Distillation","date":"2020-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/multi-level-knowledge-distillation","slug":"multi-level-knowledge-distillation","title":"Multi-level Knowledge Distillation via Knowledge Alignment and Correlation","date":"2020-12-01","arxiv_id":"2012.00573","n_code_links":1,"syntology":null},{"paper":null,"slug":"query-distillation-bert-based-distillation","title":"Query Distillation: BERT-based Distillation for Ensemble Ranking","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"reverse-engineering-recurrent-neural-network","title":"Reverse-engineering recurrent neural network solutions to a hierarchical inference task for mice","date":"2020-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"solvable-model-for-inheriting-the","title":"Solvable Model for Inheriting the Regularization through Knowledge Distillation","date":"2020-12-01","arxiv_id":"2012.00194","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-selective-survey-on-versatile-knowledge","title":"A Selective Survey on Versatile Knowledge Distillation Paradigm for Neural Network Models","date":"2020-11-30","arxiv_id":"2011.14554","n_code_links":0,"syntology":null},{"paper":"/paper/kd-lib-a-pytorch-library-for-knowledge","slug":"kd-lib-a-pytorch-library-for-knowledge","title":"KD-Lib: A PyTorch library for Knowledge Distillation, Pruning and Quantization","date":"2020-11-30","arxiv_id":"2011.14691","n_code_links":1,"syntology":null},{"paper":null,"slug":"real-time-spatio-temporal-action-localization","title":"Real-time Spatio-temporal Action Localization via Learning Motion Representation","date":"2020-11-30","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"adaptive-multiplane-image-generation-from-a","title":"Adaptive Multiplane Image Generation from a Single Internet Picture","date":"2020-11-26","arxiv_id":"2011.13317","n_code_links":0,"syntology":null},{"paper":"/paper/torchdistill-a-modular-configuration-driven","slug":"torchdistill-a-modular-configuration-driven","title":"torchdistill: A Modular, Configuration-Driven Framework for Knowledge Distillation","date":"2020-11-25","arxiv_id":"2011.12913","n_code_links":1,"syntology":{"ran":10,"of":26,"n_ran_checked":0,"n_instrument":10,"unverified":16,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 10 where Syntology's instrument failed) · 16 unverified","official":{"repos":["yoshitomo-matsubara/torchdistill"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":16,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"generative-adversarial-simulator","title":"Generative Adversarial Simulator","date":"2020-11-23","arxiv_id":"2011.11472","n_code_links":0,"syntology":null},{"paper":"/paper/evolving-search-space-for-neural-architecture","slug":"evolving-search-space-for-neural-architecture","title":"Evolving Search Space for Neural Architecture Search","date":"2020-11-22","arxiv_id":"2011.10904","n_code_links":1,"syntology":null},{"paper":"/paper/head-network-distillation-splitting-distilled","slug":"head-network-distillation-splitting-distilled","title":"Head Network Distillation: Splitting Distilled Deep Neural Networks for Resource-Constrained Edge Computing Systems","date":"2020-11-20","arxiv_id":null,"n_code_links":2,"syntology":null},{"paper":"/paper/kd3a-unsupervised-multi-source-decentralized","slug":"kd3a-unsupervised-multi-source-decentralized","title":"KD3A: Unsupervised Multi-Source Decentralized Domain Adaptation via Knowledge Distillation","date":"2020-11-19","arxiv_id":"2011.09757","n_code_links":1,"syntology":null},{"paper":null,"slug":"effectiveness-of-arbitrary-transfer-sets-for","title":"Effectiveness of Arbitrary Transfer Sets for Data-free Knowledge Distillation","date":"2020-11-18","arxiv_id":"2011.09113","n_code_links":0,"syntology":null},{"paper":"/paper/privileged-knowledge-distillation-for-online","slug":"privileged-knowledge-distillation-for-online","title":"Privileged Knowledge Distillation for Online Action Detection","date":"2020-11-18","arxiv_id":"2011.09158","n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-serial-number-computational-watermarking","title":"Deep Serial Number: Computational Watermarking for DNN Intellectual Property Protection","date":"2020-11-17","arxiv_id":"2011.08960","n_code_links":0,"syntology":null},{"paper":null,"slug":"generalized-continual-zero-shot-learning","title":"Generalized Continual Zero-Shot Learning","date":"2020-11-17","arxiv_id":"2011.08508","n_code_links":0,"syntology":null},{"paper":"/paper/anomaly-detection-in-video-via-self","slug":"anomaly-detection-in-video-via-self","title":"Anomaly Detection in Video via Self-Supervised and Multi-Task Learning","date":"2020-11-15","arxiv_id":"2011.07491","n_code_links":1,"syntology":null},{"paper":"/paper/online-ensemble-model-compression-using-1","slug":"online-ensemble-model-compression-using-1","title":"Online Ensemble Model Compression using Knowledge Distillation","date":"2020-11-15","arxiv_id":"2011.07449","n_code_links":1,"syntology":null},{"paper":"/paper/distill2vec-dynamic-graph-representation","slug":"distill2vec-dynamic-graph-representation","title":"Distill2Vec: Dynamic Graph Representation Learning with Knowledge Distillation","date":"2020-11-11","arxiv_id":"2011.05664","n_code_links":1,"syntology":null},{"paper":"/paper/egad-evolving-graph-representation-learning","slug":"egad-evolving-graph-representation-learning","title":"EGAD: Evolving Graph Representation Learning with Self-Attention and Knowledge Distillation for Live Video Streaming Events","date":"2020-11-11","arxiv_id":"2011.05705","n_code_links":1,"syntology":null},{"paper":"/paper/real-time-decentralized-knowledge-transfer-at","slug":"real-time-decentralized-knowledge-transfer-at","title":"Real-Time Decentralized knowledge Transfer at the Edge","date":"2020-11-11","arxiv_id":"2011.05961","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-estimating-the-training-cost-of","title":"On Estimating the Training Cost of Conversational Recommendation Systems","date":"2020-11-10","arxiv_id":"2011.05302","n_code_links":0,"syntology":null},{"paper":"/paper/knowledge-distillation-for-singing-voice","slug":"knowledge-distillation-for-singing-voice","title":"Knowledge Distillation for Singing Voice Detection","date":"2020-11-09","arxiv_id":"2011.04297","n_code_links":1,"syntology":null},{"paper":null,"slug":"ensembled-ctr-prediction-via-knowledge","title":"Ensemble Knowledge Distillation for CTR Prediction","date":"2020-11-08","arxiv_id":"2011.04106","n_code_links":0,"syntology":null},{"paper":null,"slug":"channel-planting-for-deep-neural-networks","title":"Channel Planting for Deep Neural Networks using Knowledge Distillation","date":"2020-11-04","arxiv_id":"2011.02390","n_code_links":0,"syntology":null},{"paper":"/paper/federated-knowledge-distillation","slug":"federated-knowledge-distillation","title":"Federated Knowledge Distillation","date":"2020-11-04","arxiv_id":"2011.02367","n_code_links":4,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":null}},{"paper":null,"slug":"on-self-distilling-graph-neural-network","title":"On Self-Distilling Graph Neural Network","date":"2020-11-04","arxiv_id":"2011.02255","n_code_links":0,"syntology":null},{"paper":"/paper/a-comprehensive-study-of-class-incremental","slug":"a-comprehensive-study-of-class-incremental","title":"A Comprehensive Study of Class Incremental Learning Algorithms for Visual Tasks","date":"2020-11-03","arxiv_id":"2011.01844","n_code_links":1,"syntology":null},{"paper":"/paper/unsupervised-domain-adaptive-knowledge","slug":"unsupervised-domain-adaptive-knowledge","title":"Domain Adaptive Knowledge Distillation for Driving Scene Semantic Segmentation","date":"2020-11-03","arxiv_id":"2011.08007","n_code_links":1,"syntology":null},{"paper":"/paper/data-free-knowledge-distillation-for","slug":"data-free-knowledge-distillation-for","title":"Data-free Knowledge Distillation for Segmentation using Data-Enriching GAN","date":"2020-11-02","arxiv_id":"2011.00809","n_code_links":1,"syntology":null},{"paper":null,"slug":"bridging-the-gap-between-prior-and-posterior","title":"Bridging the Gap between Prior and Posterior Knowledge Selection for Knowledge-Grounded Dialogue Generation","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-end-to-end-coreference-resolution-for","title":"Fast End-to-end Coreference Resolution for Korean","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"hw-tscs-participation-in-the-wmt-2020-news","title":"HW-TSC’s Participation in the WMT 2020 News Translation Shared Task","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"mixkd-towards-efficient-distillation-of-large-1","title":"MixKD: Towards Efficient Distillation of Large-scale Language Models","date":"2020-11-01","arxiv_id":"2011.00593","n_code_links":0,"syntology":null},{"paper":null,"slug":"the-niutrans-machine-translation-systems-for-2","title":"The NiuTrans Machine Translation Systems for WMT20","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"using-the-past-knowledge-to-improve-sentiment","title":"Using the Past Knowledge to Improve Sentiment Classification","date":"2020-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"proxylesskd-direct-knowledge-distillation-1","title":"ProxylessKD: Direct Knowledge Distillation with Inherited Classifier for Face Recognition","date":"2020-10-31","arxiv_id":"2011.00265","n_code_links":0,"syntology":null},{"paper":null,"slug":"activation-map-adaptation-for-effective","title":"Activation Map Adaptation for Effective Knowledge Distillation","date":"2020-10-26","arxiv_id":"2010.13500","n_code_links":0,"syntology":null},{"paper":null,"slug":"empowering-knowledge-distillation-via-open","title":"Empowering Knowledge Distillation via Open Set Recognition for Robust 3D Point Cloud Classification","date":"2020-10-25","arxiv_id":"2010.13114","n_code_links":0,"syntology":null},{"paper":"/paper/two-stage-textual-knowledge-distillation-to","slug":"two-stage-textual-knowledge-distillation-to","title":"Two-stage Textual Knowledge Distillation for End-to-End Spoken Language Understanding","date":"2020-10-25","arxiv_id":"2010.13105","n_code_links":1,"syntology":null},{"paper":null,"slug":"improved-synthetic-training-for-reading","title":"Improved Synthetic Training for Reading Comprehension","date":"2020-10-24","arxiv_id":"2010.12776","n_code_links":0,"syntology":null},{"paper":"/paper/pre-trained-summarization-distillation","slug":"pre-trained-summarization-distillation","title":"Pre-trained Summarization Distillation","date":"2020-10-24","arxiv_id":"2010.13002","n_code_links":1,"syntology":null},{"paper":"/paper/distilling-dense-representations-for-ranking","slug":"distilling-dense-representations-for-ranking","title":"Distilling Dense Representations for Ranking using Tightly-Coupled Teachers","date":"2020-10-22","arxiv_id":"2010.11386","n_code_links":2,"syntology":null},{"paper":null,"slug":"contextualized-attention-based-knowledge","title":"Contextualized Attention-based Knowledge Transfer for Spoken Conversational Question Answering","date":"2020-10-21","arxiv_id":"2010.11066","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-distillation-for-improved-accuracy","title":"Knowledge Distillation for Improved Accuracy in Spoken Question Answering","date":"2020-10-21","arxiv_id":"2010.11067","n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-video-salient-object-detection-via","title":"Fast Video Salient Object Detection via Spatiotemporal Knowledge Distillation","date":"2020-10-20","arxiv_id":"2010.10027","n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-distillation-in-wide-neural","title":"Knowledge Distillation in Wide Neural Networks: Risk Bound, Data Efficiency and Imperfect Teacher","date":"2020-10-20","arxiv_id":"2010.10090","n_code_links":0,"syntology":null},{"paper":null,"slug":"comparing-fisher-information-regularization","title":"Comparing Fisher Information Regularization with Distillation for DNN Quantization","date":"2020-10-19","arxiv_id":null,"n_code_links":0,"syntology":null}],"record_sha256":"1f74676f5e53b2a4577436215142cb4209e77b1d6d244f92d29a89c0c3c0b338","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}