{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/knowledge-distillation/papers/41","list_of":"/task/knowledge-distillation","task":"Knowledge Distillation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":41,"pages_in_order":43,"rows_per_page":100,"rows":[4001,4100],"of":4240,"counts":{"archive_papers_tagged":4240,"with_a_code_link":1740,"where_syntology_ran_a_sample":451,"not_listed_spam_title":0,"listed":4240,"listed_where_code_ran":451,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":380,"every_run_a_failure_of_syntologys_instrument":71,"listed_with_a_run_with_no_instrument_failure":380,"listed_every_run_a_failure_of_syntologys_instrument":71,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/knowledge-distillation","prev":"/task/knowledge-distillation/papers/40","next":"/task/knowledge-distillation/papers/42","papers":[{"url":null,"slug":"improving-autoregressive-nmt-with-non","title":"Improving Autoregressive NMT with Non-Autoregressive Model","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"simulspeech-end-to-end-simultaneous-speech-to","title":"SimulSpeech: End-to-End Simultaneous Speech to Text Translation","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"xiaomi-s-submissions-for-iwslt-2020-open","title":"Xiaomi's Submissions for IWSLT 2020 Open Domain Translation Task","date":"2020-07-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"extracurricular-learning-knowledge-transfer","title":"Extracurricular Learning: Knowledge Transfer Beyond Empirical Distribution","date":"2020-06-30","arxiv_id":"2007.00051","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-demystification-of-knowledge","title":"On the Demystification of Knowledge Distillation: A Residual Network Perspective","date":"2020-06-30","arxiv_id":"2006.16589","repositories_listed":0,"syntology":null},{"url":null,"slug":"motion-pyramid-networks-for-accurate-and","title":"Motion Pyramid Networks for Accurate and Efficient Cardiac Motion Estimation","date":"2020-06-28","arxiv_id":"2006.15710","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-diverse-latent-representations-for","title":"Diverse Knowledge Distillation (DKD): A Solution for Improving The Robustness of Ensemble Models Against Adversarial Attacks","date":"2020-06-26","arxiv_id":"2006.15127","repositories_listed":0,"syntology":null},{"url":null,"slug":"streaming-transformer-asr-with-blockwise","title":"Streaming Transformer ASR with Blockwise Synchronous Inference","date":"2020-06-25","arxiv_id":"2006.14941","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-object-detectors-with-task","title":"Distilling Object Detectors with Task Adaptive Regularization","date":"2020-06-23","arxiv_id":"2006.13108","repositories_listed":0,"syntology":null},{"url":null,"slug":"prior-knowledge-distillation-based-on","title":"Prior knowledge distillation based on financial time series","date":"2020-06-16","arxiv_id":"2006.09247","repositories_listed":0,"syntology":null},{"url":null,"slug":"pixel-invisibility-detecting-objects","title":"Pixel Invisibility: Detecting Objects Invisible in Color Images","date":"2020-06-15","arxiv_id":"2006.08383","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-a-survey","title":"Knowledge Distillation: A Survey","date":"2020-06-09","arxiv_id":"2006.05525","repositories_listed":0,"syntology":null},{"url":null,"slug":"reskd-residual-guided-knowledge-distillation","title":"ResKD: Residual-Guided Knowledge Distillation","date":"2020-06-08","arxiv_id":"2006.04719","repositories_listed":0,"syntology":null},{"url":null,"slug":"admp-an-adversarial-double-masks-based","title":"ADMP: An Adversarial Double Masks Based Pruning Framework For Unsupervised Cross-Domain Compression","date":"2020-06-07","arxiv_id":"2006.04127","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-analysis-of-the-impact-of-data","title":"An Empirical Analysis of the Impact of Data Augmentation on Knowledge Distillation","date":"2020-06-06","arxiv_id":"2006.03810","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-overview-of-neural-network-compression","title":"An Overview of Neural Network Compression","date":"2020-06-05","arxiv_id":"2006.03669","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-speech-translation-with-knowledge-1","title":"End-to-End Speech-Translation with Knowledge Distillation: FBK@IWSLT2020","date":"2020-06-04","arxiv_id":"2006.02965","repositories_listed":0,"syntology":null},{"url":null,"slug":"adinet-attribute-driven-incremental-network","title":"ADINet: Attribute Driven Incremental Network for Retinal Image Classification","date":"2020-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"apprentissage-automatique-de-repr-esentation","title":"Apprentissage automatique de repr\\'esentation de voix \\`a l'aide d'une distillation de la connaissance pour le casting vocal (Learning voice representation using knowledge distillation for automatic voice casting )","date":"2020-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"weight-squeezing-reparameterization-for-1","title":"Weight Squeezing: Reparameterization for Compression and Fast Inference","date":"2020-05-30","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sub-band-knowledge-distillation-framework-for","title":"Sub-Band Knowledge Distillation Framework for Speech Enhancement","date":"2020-05-29","arxiv_id":"2005.14435","repositories_listed":0,"syntology":null},{"url":null,"slug":"syntactic-structure-distillation-pretraining","title":"Syntactic Structure Distillation Pretraining For Bidirectional Encoders","date":"2020-05-27","arxiv_id":"2005.13482","repositories_listed":0,"syntology":null},{"url":null,"slug":"why-distillation-helps-a-statistical","title":"Why distillation helps: a statistical perspective","date":"2020-05-21","arxiv_id":"2005.10419","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-a-lightweight-teacher-for","title":"Learning from a Lightweight Teacher for Efficient Knowledge Distillation","date":"2020-05-19","arxiv_id":"2005.09163","repositories_listed":0,"syntology":null},{"url":null,"slug":"incremental-learning-for-end-to-end-automatic","title":"Incremental Learning for End-to-End Automatic Speech Recognition","date":"2020-05-11","arxiv_id":"2005.04288","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-knowledge-from-pre-trained","title":"Distilling Knowledge from Pre-trained Language Models via Text Smoothing","date":"2020-05-08","arxiv_id":"2005.03848","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-non-autoregressive-neural-machine","title":"Improving Non-autoregressive Neural Machine Translation with Monolingual Data","date":"2020-05-02","arxiv_id":"2005.00932","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-spikes-knowledge-distillation-in","title":"Distilling Spikes: Knowledge Distillation in Spiking Neural Networks","date":"2020-05-01","arxiv_id":"2005.00288","repositories_listed":0,"syntology":null},{"url":null,"slug":"general-purpose-text-embeddings-from-pre","title":"General Purpose Text Embeddings from Pre-trained Language Models for Scalable Inference","date":"2020-04-29","arxiv_id":"2004.14287","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightpaff-a-two-stage-distillation-framework-1","title":"LightPAFF: A Two-Stage Distillation Framework for Pre-training and Fine-tuning","date":"2020-04-27","arxiv_id":"2004.12817","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-study-of-non-autoregressive-model-for","title":"A Study of Non-autoregressive Model for Sequence Generation","date":"2020-04-22","arxiv_id":"2004.10454","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-for-multilingual","title":"Knowledge Distillation for Multilingual Unsupervised Neural Machine Translation","date":"2020-04-21","arxiv_id":"2004.10171","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-for-action","title":"Knowledge Distillation for Action Anticipation via Label Smoothing","date":"2020-04-16","arxiv_id":"2004.07711","repositories_listed":0,"syntology":null},{"url":null,"slug":"building-a-multi-domain-neural-machine","title":"Building a Multi-domain Neural Machine Translation Model using Knowledge Distillation","date":"2020-04-15","arxiv_id":"2004.07324","repositories_listed":0,"syntology":null},{"url":null,"slug":"smart-inference-for-multidigit-convolutional","title":"Smart Inference for Multidigit Convolutional Neural Network based Barcode Decoding","date":"2020-04-14","arxiv_id":"2004.06297","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-robust-classification-with-image","title":"Towards Robust Classification with Image Quality Assessment","date":"2020-04-14","arxiv_id":"2004.06288","repositories_listed":0,"syntology":null},{"url":null,"slug":"tinymbert-multi-stage-distillation-framework","title":"XtremeDistil: Multi-stage Distillation for Massive Multilingual Models","date":"2020-04-12","arxiv_id":"2004.05686","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-for-mobile-edge","title":"Knowledge Distillation for Mobile Edge Computation Offloading","date":"2020-04-09","arxiv_id":"2004.04366","repositories_listed":0,"syntology":null},{"url":null,"slug":"ladabert-lightweight-adaptation-of-bert","title":"LadaBERT: Lightweight Adaptation of BERT through Hybrid Model Compression","date":"2020-04-08","arxiv_id":"2004.04124","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-non-task-specific-distillation-of","title":"Towards Non-task-specific Distillation of BERT via Sentence Representation Approximation","date":"2020-04-07","arxiv_id":"2004.03097","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-review-comprehension-with-domain","title":"Enhancing Review Comprehension with Domain-Specific Commonsense","date":"2020-04-06","arxiv_id":"2004.03020","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-as-priors-cross-modal-knowledge","title":"Knowledge as Priors: Cross-Modal Knowledge Generalization for Datasets without Superior Knowledge","date":"2020-04-01","arxiv_id":"2004.00176","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simple-class-decision-balancing-for","title":"SS-IL: Separated Softmax for Incremental Learning","date":"2020-03-31","arxiv_id":"2003.13947","repositories_listed":0,"syntology":null},{"url":null,"slug":"spatio-temporal-graph-for-video-captioning","title":"Spatio-Temporal Graph for Video Captioning with Knowledge Distillation","date":"2020-03-31","arxiv_id":"2003.13942","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-unreasonable-effectiveness-of","title":"Analysis of Knowledge Transfer in Kernel Regime","date":"2020-03-30","arxiv_id":"2003.13438","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-methods-for-low-power-deep","title":"A Survey of Methods for Low-Power Deep Learning and Computer Vision","date":"2020-03-24","arxiv_id":"2003.11066","repositories_listed":0,"syntology":null},{"url":null,"slug":"synergic-adversarial-label-learning-with-dr","title":"Synergic Adversarial Label Learning for Grading Retinal Diseases via Knowledge Distillation and Multi-task Learning","date":"2020-03-24","arxiv_id":"2003.10607","repositories_listed":0,"syntology":null},{"url":null,"slug":"teacher-student-chain-for-efficient-semi","title":"Teacher-Student chain for efficient semi-supervised histology image classification","date":"2020-03-17","arxiv_id":"2003.08797","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-via-adaptive-instance","title":"Knowledge distillation via adaptive instance normalization","date":"2020-03-09","arxiv_id":"2003.04289","repositories_listed":0,"syntology":null},{"url":null,"slug":"pacemaker-intermediate-teacher-knowledge","title":"Pacemaker: Intermediate Teacher Knowledge Distillation For On-The-Fly Convolutional Neural Network","date":"2020-03-09","arxiv_id":"2003.03944","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-portable-generative-adversarial","title":"Distilling portable Generative Adversarial Networks for Image Translation","date":"2020-03-07","arxiv_id":"2003.03519","repositories_listed":0,"syntology":null},{"url":null,"slug":"explaining-knowledge-distillation-by","title":"Explaining Knowledge Distillation by Quantifying the Knowledge","date":"2020-03-07","arxiv_id":"2003.03622","repositories_listed":0,"syntology":null},{"url":null,"slug":"distill-adapt-distill-training-small-in","title":"Distill, Adapt, Distill: Training Small, In-Domain Models for Neural Machine Translation","date":"2020-03-05","arxiv_id":"2003.02877","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-efficient-method-of-training-small-models","title":"An Efficient Method of Training Small Models for Regression Problems with Knowledge Distillation","date":"2020-02-28","arxiv_id":"2002.12597","repositories_listed":0,"syntology":null},{"url":null,"slug":"residual-knowledge-distillation","title":"Residual Knowledge Distillation","date":"2020-02-21","arxiv_id":"2002.09168","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-cost-and-benefit-with-tied-multi-1","title":"Balancing Cost and Benefit with Tied-Multi Transformers","date":"2020-02-20","arxiv_id":"2002.08614","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-distillation-amplifies-regularization-in","title":"Self-Distillation Amplifies Regularization in Hilbert Space","date":"2020-02-13","arxiv_id":"2002.05715","repositories_listed":0,"syntology":null},{"url":null,"slug":"meta-learning-across-meta-tasks-for-few-shot","title":"Meta-Learning across Meta-Tasks for Few-Shot Learning","date":"2020-02-11","arxiv_id":"2002.04274","repositories_listed":0,"syntology":null},{"url":null,"slug":"population-based-training-for-loss-function","title":"Regularized Evolutionary Population-Based Training","date":"2020-02-11","arxiv_id":"2002.04225","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-and-improving-knowledge","title":"Understanding and Improving Knowledge Distillation","date":"2020-02-10","arxiv_id":"2002.03532","repositories_listed":0,"syntology":null},{"url":null,"slug":"unlabeled-data-deployment-for-classification","title":"Unlabeled Data Deployment for Classification of Diabetic Retinopathy Images Using Knowledge Transfer","date":"2020-02-09","arxiv_id":"2002.03321","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-map-level-online-adversarial-1","title":"Feature-map-level Online Adversarial Knowledge Distillation","date":"2020-02-05","arxiv_id":"2002.01775","repositories_listed":0,"syntology":null},{"url":null,"slug":"search-for-better-students-to-learn-distilled","title":"Search for Better Students to Learn Distilled Knowledge","date":"2020-01-30","arxiv_id":"2001.11612","repositories_listed":0,"syntology":null},{"url":null,"slug":"developing-multi-task-recommendations-with","title":"Developing Multi-Task Recommendations with Long-Term Rewards via Policy Distilled Reinforcement Learning","date":"2020-01-27","arxiv_id":"2001.09595","repositories_listed":0,"syntology":null},{"url":null,"slug":"generation-distillation-for-efficient-natural-1","title":"Generation-Distillation for Efficient Natural Language Understanding in Low-Data Settings","date":"2020-01-25","arxiv_id":"2002.00733","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-techniques-for-online-end-to-end-speech","title":"Data Techniques For Online End-to-end Speech Recognition","date":"2020-01-24","arxiv_id":"2001.09221","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-network-pruning-network-approach-to-deep","title":"A \"Network Pruning Network\" Approach to Deep Model Compression","date":"2020-01-15","arxiv_id":"2001.05545","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-3d-human-pose-estimation-network","title":"Lightweight 3D Human Pose Estimation Network Training Using Teacher-Student Learning","date":"2020-01-15","arxiv_id":"2001.05097","repositories_listed":0,"syntology":null},{"url":null,"slug":"uncertainty-aware-multi-shot-knowledge","title":"Uncertainty-Aware Multi-Shot Knowledge Distillation for Image-Based Object Re-Identification","date":"2020-01-15","arxiv_id":"2001.05197","repositories_listed":0,"syntology":null},{"url":null,"slug":"noisy-machines-understanding-noisy-neural-1","title":"Noisy Machines: Understanding Noisy Neural Networks and Enhancing Robustness to Analog Hardware Errors Using Distillation","date":"2020-01-14","arxiv_id":"2001.04974","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-task-agnostic-embedding-of-multiple","title":"Learning Task-Agnostic Embedding of Multiple Black-Box Experts for Multi-Task Model Fusion","date":"2020-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-teacher-student-techniques-in-deep","title":"Modeling Teacher-Student Techniques in Deep Neural Networks for Knowledge Distillation","date":"2019-12-31","arxiv_id":"1912.13179","repositories_listed":0,"syntology":null},{"url":null,"slug":"degan-data-enriching-gan-for-retrieving","title":"DeGAN : Data-Enriching GAN for Retrieving Representative Samples from a Trained Classifier","date":"2019-12-27","arxiv_id":"1912.11960","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-architecture-and-knowledge-distillation","title":"Joint Architecture and Knowledge Distillation in CNN for Chinese Text Recognition","date":"2019-12-17","arxiv_id":"1912.07806","repositories_listed":0,"syntology":null},{"url":null,"slug":"iterative-dual-domain-adaptation-for-neural-1","title":"Iterative Dual Domain Adaptation for Neural Machine Translation","date":"2019-12-16","arxiv_id":"1912.07239","repositories_listed":0,"syntology":null},{"url":null,"slug":"explaining-sequence-level-knowledge","title":"Explaining Sequence-Level Knowledge Distillation as Data-Augmentation for Neural Machine Translation","date":"2019-12-06","arxiv_id":"1912.03334","repositories_listed":0,"syntology":null},{"url":null,"slug":"acquiring-knowledge-from-pre-trained-model-to","title":"Acquiring Knowledge from Pre-trained Model to Neural Machine Translation","date":"2019-12-04","arxiv_id":"1912.01774","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-convolutional-neural-networks-for-2","title":"Efficient Convolutional Neural Networks for Depth-Based Multi-Person Pose Estimation","date":"2019-12-02","arxiv_id":"1912.00711","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-oracle-knowledge-distillation-with","title":"Towards Oracle Knowledge Distillation with Neural Architecture Search","date":"2019-11-29","arxiv_id":"1911.13019","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-driven-compression-of-convolutional","title":"Data-Driven Compression of Convolutional Neural Networks","date":"2019-11-28","arxiv_id":"1911.12740","repositories_listed":0,"syntology":null},{"url":null,"slug":"qkd-quantization-aware-knowledge-distillation","title":"QKD: Quantization-aware Knowledge Distillation","date":"2019-11-28","arxiv_id":"1911.12491","repositories_listed":0,"syntology":null},{"url":null,"slug":"search-to-distill-pearls-are-everywhere-but","title":"Search to Distill: Pearls are Everywhere but not the Eyes","date":"2019-11-20","arxiv_id":"1911.09074","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-making-deep-transfer-learning-never","title":"Towards Making Deep Transfer Learning Never Hurt","date":"2019-11-18","arxiv_id":"1911.07489","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-distillation-for-top-n","title":"Collaborative Distillation for Top-N Recommendation","date":"2019-11-13","arxiv_id":"1911.05276","repositories_listed":0,"syntology":null},{"url":"/paper/knowledge-representing-efficient-sparse","slug":"knowledge-representing-efficient-sparse","title":"Knowledge Representing: Efficient, Sparse Representation of Prior Knowledge for Knowledge Distillation","date":"2019-11-13","arxiv_id":"1911.05329","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-representation-learning-via-multi-task","title":"Graph Representation Learning via Multi-task Knowledge Distillation","date":"2019-11-11","arxiv_id":"1911.05700","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-in-document-retrieval","title":"Knowledge Distillation in Document Retrieval","date":"2019-11-11","arxiv_id":"1911.11065","repositories_listed":0,"syntology":null},{"url":null,"slug":"attentive-student-meets-multi-task-teacher","title":"MKD: a Multi-Task Knowledge Distillation Approach for Pretrained Language Models","date":"2019-11-09","arxiv_id":"1911.03588","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-for-incremental","title":"Knowledge Distillation for Incremental Learning in Semantic Segmentation","date":"2019-11-08","arxiv_id":"1911.03462","repositories_listed":0,"syntology":null},{"url":"/paper/microsoft-research-asias-systems-for-wmt19-1","slug":"microsoft-research-asias-systems-for-wmt19-1","title":"Microsoft Research Asia's Systems for WMT19","date":"2019-11-07","arxiv_id":"1911.06191","repositories_listed":0,"syntology":null},{"url":null,"slug":"teacher-student-training-for-robust-tacotron","title":"Teacher-Student Training for Robust Tacotron-based TTS","date":"2019-11-07","arxiv_id":"1911.02839","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-knowledge-distillation-in-non","title":"Understanding Knowledge Distillation in Non-autoregressive Machine Translation","date":"2019-11-07","arxiv_id":"1911.02727","repositories_listed":0,"syntology":null},{"url":null,"slug":"espnet-how2-speech-translation-system-for","title":"ESPnet How2 Speech Translation System for IWSLT 2019: Pre-training, Knowledge Distillation, and Going Deeper","date":"2019-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-cross-lingual-semantic","title":"Weakly Supervised Cross-lingual Semantic Relation Classification via Knowledge Distillation","date":"2019-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-pixel-wise-feature-similarities","title":"Distilling Pixel-Wise Feature Similarities for Semantic Segmentation","date":"2019-10-31","arxiv_id":"1910.14226","repositories_listed":0,"syntology":null},{"url":"/paper/a-simple-but-effective-bert-model-for-dialog","slug":"a-simple-but-effective-bert-model-for-dialog","title":"A Simple but Effective BERT Model for Dialog State Tracking on Resource-Limited Systems","date":"2019-10-28","arxiv_id":"1910.12995","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-student-learning-compact-and","title":"Variational Student: Learning Compact and Sparser Networks in Knowledge Distillation Framework","date":"2019-10-26","arxiv_id":"1910.12061","repositories_listed":0,"syntology":null},{"url":null,"slug":"secost-sequential-co-supervision-for-weakly","title":"Secost: Sequential co-supervision for large scale weakly labeled audio event detection","date":"2019-10-25","arxiv_id":"1910.11789","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarial-feature-alignment-avoid","title":"Adversarial Feature Alignment: Avoid Catastrophic Forgetting in Incremental Task Lifelong Learning","date":"2019-10-24","arxiv_id":"1910.10986","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-empirical-study-of-efficient-asr-rescoring","title":"An Empirical Study of Efficient ASR Rescoring with Transformers","date":"2019-10-24","arxiv_id":"1910.11450","repositories_listed":0,"syntology":null}],"record_sha256":"29c251cc25182b42a07d50814e31f8a9eb8653bce52c6d1d53eb935e288d8da8","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}