{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/knowledge-distillation/papers/39","list_of":"/task/knowledge-distillation","task":"Knowledge Distillation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":39,"pages_in_order":43,"rows_per_page":100,"rows":[3801,3900],"of":4240,"counts":{"archive_papers_tagged":4240,"with_a_code_link":1740,"where_syntology_ran_a_sample":451,"not_listed_spam_title":0,"listed":4240,"listed_where_code_ran":451,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":380,"every_run_a_failure_of_syntologys_instrument":71,"listed_with_a_run_with_no_instrument_failure":380,"listed_every_run_a_failure_of_syntologys_instrument":71,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/knowledge-distillation","prev":"/task/knowledge-distillation/papers/38","next":"/task/knowledge-distillation/papers/40","papers":[{"url":"/paper/transformer-based-asr-incorporating-time","slug":"transformer-based-asr-incorporating-time","title":"Transformer-based ASR Incorporating Time-reduction Layer and Fine-tuning with Self-Knowledge Distillation","date":"2021-03-17","arxiv_id":"2103.09903","repositories_listed":0,"syntology":null},{"url":"/paper/leveraging-recent-advances-in-deep-learning","slug":"leveraging-recent-advances-in-deep-learning","title":"Leveraging Recent Advances in Deep Learning for Audio-Visual Emotion Recognition","date":"2021-03-16","arxiv_id":"2103.09154","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustly-optimized-and-distilled-training-for","title":"Robustly Optimized and Distilled Training for Natural Language Understanding","date":"2021-03-16","arxiv_id":"2103.08809","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-feature-regularization-self-feature","title":"A New Training Framework for Deep Neural Network","date":"2021-03-12","arxiv_id":"2103.07350","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-aware-knowledge-distillation-for-few","title":"Semantic-aware Knowledge Distillation for Few-Shot Class-Incremental Learning","date":"2021-03-06","arxiv_id":"2103.04059","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-neural-network-models-compression","title":"Deep Neural Network Models Compression","date":"2021-03-04","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-align-network-and-knowledge","title":"Feature-Align Network with Knowledge Distillation for Efficient Denoising","date":"2021-03-02","arxiv_id":"2103.01524","repositories_listed":0,"syntology":null},{"url":null,"slug":"embedded-knowledge-distillation-in-depth","title":"Embedded Knowledge Distillation in Depth-Level Dynamic Neural Network","date":"2021-03-01","arxiv_id":"2103.00793","repositories_listed":0,"syntology":null},{"url":null,"slug":"alignment-knowledge-distillation-for-online","title":"Alignment Knowledge Distillation for Online Streaming Attention-based Speech Recognition","date":"2021-02-28","arxiv_id":"2103.00422","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-circumvents","title":"Knowledge Distillation Circumvents Nonlinearity for Optical Convolutional Neural Networks","date":"2021-02-26","arxiv_id":"2102.13323","repositories_listed":0,"syntology":null},{"url":null,"slug":"pursuhint-in-search-of-informative-hint","title":"PURSUhInT: In Search of Informative Hint Points Based on Layer Clustering for Knowledge Distillation","date":"2021-02-26","arxiv_id":"2103.00053","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-data-free-adversarial-distillation","title":"Enhancing Data-Free Adversarial Distillation with Activation Regularization and Virtual Interpolation","date":"2021-02-23","arxiv_id":"2102.11638","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-view-feature-representation-for","title":"Multi-View Feature Representation for Dialogue Generation with Bidirectional Distillation","date":"2021-02-22","arxiv_id":"2102.10780","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-knowledge-distillation-of-a-deep","title":"Exploring Knowledge Distillation of a Deep Neural Network for Multi-Script identification","date":"2021-02-20","arxiv_id":"2102.10335","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-automatic-speech-recognition-with","title":"End-to-End Automatic Speech Recognition with Deep Mutual Learning","date":"2021-02-16","arxiv_id":"2102.08154","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-transformer-based-large-context","title":"Hierarchical Transformer-based Large-Context End-to-end ASR with Large-Context Knowledge Distillation","date":"2021-02-16","arxiv_id":"2102.07935","repositories_listed":0,"syntology":null},{"url":null,"slug":"cap-gan-towards-adversarial-robustness-with","title":"CAP-GAN: Towards Adversarial Robustness with Cycle-consistent Attentional Purification","date":"2021-02-15","arxiv_id":"2102.07304","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-customer-transaction-classification","title":"Improved Customer Transaction Classification using Semi-Supervised Knowledge Distillation","date":"2021-02-15","arxiv_id":"2102.07635","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-acoustic-and-linguistic-embeddings","title":"Leveraging Acoustic and Linguistic Embeddings from Pretrained speech and language Models for Intent Classification","date":"2021-02-15","arxiv_id":"2102.07370","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-regulated-learning-mechanism-for-data","title":"Self Regulated Learning Mechanism for Data Efficient Knowledge Distillation","date":"2021-02-14","arxiv_id":"2102.07125","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-student-friendly-teacher-networks","title":"Learning Student-Friendly Teacher Networks for Knowledge Distillation","date":"2021-02-12","arxiv_id":"2102.07650","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantically-conditioned-negative-samples-for","title":"Semantically-Conditioned Negative Samples for Efficient Contrastive Learning","date":"2021-02-12","arxiv_id":"2102.06603","repositories_listed":0,"syntology":null},{"url":null,"slug":"newsbert-distilling-pre-trained-language","title":"NewsBERT: Distilling Pre-trained Language Model for Intelligent News Application","date":"2021-02-09","arxiv_id":"2102.04887","repositories_listed":0,"syntology":null},{"url":null,"slug":"do-not-forget-to-attend-to-uncertainty-while","title":"Do Not Forget to Attend to Uncertainty while Mitigating Catastrophic Forgetting","date":"2021-02-03","arxiv_id":"2102.01906","repositories_listed":0,"syntology":null},{"url":null,"slug":"isp-distillation","title":"ISP Distillation","date":"2021-01-25","arxiv_id":"2101.10203","repositories_listed":0,"syntology":null},{"url":null,"slug":"network-agnostic-knowledge-transfer-for","title":"Network-Agnostic Knowledge Transfer for Medical Image Segmentation","date":"2021-01-23","arxiv_id":"2101.09560","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-human-action","title":"Bridging the gap between Human Action Recognition and Online Action Detection","date":"2021-01-21","arxiv_id":"2101.08851","repositories_listed":0,"syntology":null},{"url":null,"slug":"collaborative-teacher-student-learning-via","title":"Collaborative Teacher-Student Learning via Multiple Knowledge Transfer","date":"2021-01-21","arxiv_id":"2101.08471","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-epidemiological-modeling-by-black-box","title":"Deep Epidemiological Modeling by Black-box Knowledge Distillation: An Accurate Deep Learning Model for COVID-19","date":"2021-01-20","arxiv_id":"2101.10280","repositories_listed":0,"syntology":null},{"url":"/paper/learning-to-augment-for-data-scarce-domain","slug":"learning-to-augment-for-data-scarce-domain","title":"Learning to Augment for Data-Scarce Domain BERT Knowledge Distillation","date":"2021-01-20","arxiv_id":"2101.08106","repositories_listed":0,"syntology":{"n":15,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":13,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":15,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 13 unverified","sample_list":"/paper/learning-to-augment-for-data-scarce-domain#ran","syntology_url":"https://syntology.ai/paper/2101.08106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2101.08106"}},"official":null}},{"url":null,"slug":"incremental-knowledge-based-question","title":"Incremental Knowledge Based Question Answering","date":"2021-01-18","arxiv_id":"2101.06938","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-methods-for-efficient","title":"Knowledge Distillation Methods for Efficient Unsupervised Adaptation Across Multiple Domains","date":"2021-01-18","arxiv_id":"2101.07308","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-impressions-mining-deep-models-to","title":"Mining Data Impressions from Deep Models as Substitute for the Unavailable Training Data","date":"2021-01-15","arxiv_id":"2101.06069","repositories_listed":0,"syntology":null},{"url":null,"slug":"kdlsq-bert-a-quantized-bert-combining","title":"KDLSQ-BERT: A Quantized Bert Combining Knowledge Distillation with Learned Step Size Quantization","date":"2021-01-15","arxiv_id":"2101.05938","repositories_listed":0,"syntology":null},{"url":null,"slug":"interpretable-discovery-of-new-semiconductors","title":"Interpretable discovery of new semiconductors with machine learning","date":"2021-01-12","arxiv_id":"2101.04383","repositories_listed":0,"syntology":null},{"url":null,"slug":"resolution-based-distillation-for-efficient","title":"Resolution-Based Distillation for Efficient Histology Image Classification","date":"2021-01-11","arxiv_id":"2101.04170","repositories_listed":0,"syntology":null},{"url":null,"slug":"adversarially-robust-and-explainable-model","title":"Adversarially Robust and Explainable Model Compression with On-Device Personalization for Text Classification","date":"2021-01-10","arxiv_id":"2101.05624","repositories_listed":0,"syntology":null},{"url":null,"slug":"modality-specific-distillation","title":"MSD: Saliency-aware Knowledge Distillation for Multimodal Understanding","date":"2021-01-06","arxiv_id":"2101.01881","repositories_listed":0,"syntology":null},{"url":null,"slug":"label-augmentation-via-time-based-knowledge","title":"Label Augmentation via Time-based Knowledge Distillation for Financial Anomaly Detection","date":"2021-01-05","arxiv_id":"2101.01689","repositories_listed":0,"syntology":null},{"url":null,"slug":"active-learning-for-lane-detection-a","title":"Active Learning for Lane Detection: A Knowledge Distillation Approach","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"can-students-outperform-teachers-in-knowledge","title":"Can Students Outperform Teachers in Knowledge Distillation based Model Compression?","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-knowledge-distillation-for","title":"Contextual Knowledge Distillation for Transformer Compression","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"disentanglement-visualization-and-analysis-of","title":"Disentanglement, Visualization and Analysis of Complex Features in DNNs","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"don-t-be-picky-all-students-in-the-right","title":"Don't be picky, all students in the right family can learn from good teachers","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"explicit-connection-distillation","title":"Explicit Connection Distillation","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"flar-a-unified-prototype-framework-for-few","title":"FLAR: A Unified Prototype Framework for Few-Sample Lifelong Active Recognition","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-de-raining-generalization-via","title":"Improving De-Raining Generalization via Neural Reorganization","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"kernel-methods-in-hyperbolic-spaces","title":"Kernel Methods in Hyperbolic Spaces","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-based-ensemble","title":"Knowledge Distillation based Ensemble Learning for Neural Machine Translation","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-via-softmax-regression","title":"Knowledge distillation via softmax regression representation learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-deep-model-via-exploring-local","title":"Learning from deep model via exploring local targets","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"long-live-the-lottery-the-existence-of","title":"Long Live the Lottery: The Existence of Winning Tickets in Lifelong Learning","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-soft-labels-for-knowledge","title":"Rethinking Soft Labels for Knowledge Distillation: A Bias–Variance Tradeoff Perspective","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-overfitting-may-be-mitigated-by","title":"Robust Overfitting may be mitigated by properly learned smoothening","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"student-customized-knowledge-distillation","title":"Student Customized Knowledge Distillation: Bridging the Gap Between Student and Teacher","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-adversarial-attacks-on","title":"Understanding Adversarial Attacks on Autoencoders","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-knowledge-distillation","title":"Understanding Knowledge Distillation","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unpaired-learning-for-deep-image-deraining","title":"Unpaired Learning for Deep Image Deraining With Rain Direction Regularizer","date":"2021-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-monolingual-data-for-neural-machine","title":"Fully Synthetic Data Improves Neural Machine Translation with Knowledge Distillation","date":"2020-12-31","arxiv_id":"2012.15455","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-zero-shot-knowledge-distillation-for","title":"Towards Zero-Shot Knowledge Distillation for Natural Language Processing","date":"2020-12-31","arxiv_id":"2012.15495","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-with-adaptive","title":"Knowledge Distillation with Adaptive Asymmetric Label Sharpening for Semi-supervised Fracture Detection in Chest X-rays","date":"2020-12-30","arxiv_id":"2012.15359","repositories_listed":0,"syntology":null},{"url":null,"slug":"understanding-and-improving-lexical-choice-in-1","title":"Understanding and Improving Lexical Choice in Non-Autoregressive Translation","date":"2020-12-29","arxiv_id":"2012.14583","repositories_listed":0,"syntology":null},{"url":null,"slug":"alp-kd-attention-based-layer-projection-for","title":"ALP-KD: Attention-Based Layer Projection for Knowledge Distillation","date":"2020-12-27","arxiv_id":"2012.14022","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-a-universal-continuous-knowledge-base","title":"Towards a Universal Continuous Knowledge Base","date":"2020-12-25","arxiv_id":"2012.13568","repositories_listed":0,"syntology":null},{"url":null,"slug":"future-guided-incremental-transformer-for","title":"Future-Guided Incremental Transformer for Simultaneous Translation","date":"2020-12-23","arxiv_id":"2012.12465","repositories_listed":0,"syntology":null},{"url":null,"slug":"attentionlite-towards-efficient-self","title":"AttentionLite: Towards Efficient Self-Attention Models for Vision","date":"2020-12-21","arxiv_id":"2101.05216","repositories_listed":0,"syntology":null},{"url":null,"slug":"diverse-knowledge-distillation-for-end-to-end","title":"Diverse Knowledge Distillation for End-to-End Person Search","date":"2020-12-21","arxiv_id":"2012.11187","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-ensemble-knowledge","title":"Towards Understanding Ensemble, Knowledge Distillation and Self-Distillation in Deep Learning","date":"2020-12-17","arxiv_id":"2012.09816","repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-optimal-neural-networks-rapid","title":"Distilling Optimal Neural Networks: Rapid Search in Diverse Spaces","date":"2020-12-16","arxiv_id":"2012.08859","repositories_listed":0,"syntology":null},{"url":"/paper/wasserstein-contrastive-representation","slug":"wasserstein-contrastive-representation","title":"Wasserstein Contrastive Representation Distillation","date":"2020-12-15","arxiv_id":"2012.08674","repositories_listed":0,"syntology":null},{"url":null,"slug":"lrc-bert-latent-representation-contrastive","title":"LRC-BERT: Latent-representation Contrastive Knowledge Distillation for Natural Language Understanding","date":"2020-12-14","arxiv_id":"2012.07335","repositories_listed":0,"syntology":null},{"url":null,"slug":"periocular-in-the-wild-embedding-learning","title":"Periocular Embedding Learning with Consistent Knowledge Distillation from Face","date":"2020-12-12","arxiv_id":"2012.06746","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-task-agnostic-bert-distillation","title":"Improving Task-Agnostic BERT Distillation with Layer Mapping Search","date":"2020-12-11","arxiv_id":"2012.06153","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-multi-teacher-selection-for","title":"Reinforced Multi-Teacher Selection for Knowledge Distillation","date":"2020-12-11","arxiv_id":"2012.06048","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-scale-generative-data-free-distillation","title":"Large-Scale Generative Data-Free Distillation","date":"2020-12-10","arxiv_id":"2012.05578","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-compression-using-optimal-transport","title":"Model Compression Using Optimal Transport","date":"2020-12-07","arxiv_id":"2012.03907","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-head-knowledge-distillation-for-model","title":"Multi-head Knowledge Distillation for Model Compression","date":"2020-12-05","arxiv_id":"2012.02911","repositories_listed":0,"syntology":null},{"url":null,"slug":"classification-under-misspecification-1","title":"Classification Under Misspecification: Halfspaces, Generalized Linear Models, and Evolvability","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"query-distillation-bert-based-distillation","title":"Query Distillation: BERT-based Distillation for Ensemble Ranking","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"reverse-engineering-recurrent-neural-network","title":"Reverse-engineering recurrent neural network solutions to a hierarchical inference task for mice","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"self-supervised-generative-adversarial-1","title":"Self-Supervised Generative Adversarial Compression","date":"2020-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"solvable-model-for-inheriting-the","title":"Solvable Model for Inheriting the Regularization through Knowledge Distillation","date":"2020-12-01","arxiv_id":"2012.00194","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-selective-survey-on-versatile-knowledge","title":"A Selective Survey on Versatile Knowledge Distillation Paradigm for Neural Network Models","date":"2020-11-30","arxiv_id":"2011.14554","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-spatio-temporal-action-localization","title":"Real-time Spatio-temporal Action Localization via Learning Motion Representation","date":"2020-11-30","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-multiplane-image-generation-from-a","title":"Adaptive Multiplane Image Generation from a Single Internet Picture","date":"2020-11-26","arxiv_id":"2011.13317","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-adversarial-simulator","title":"Generative Adversarial Simulator","date":"2020-11-23","arxiv_id":"2011.11472","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-in-school-multi-teacher-knowledge","title":"MixMix: All You Need for Data-Free Compression Are Feature and Data Mixing","date":"2020-11-19","arxiv_id":"2011.09899","repositories_listed":0,"syntology":null},{"url":null,"slug":"effectiveness-of-arbitrary-transfer-sets-for","title":"Effectiveness of Arbitrary Transfer Sets for Data-free Knowledge Distillation","date":"2020-11-18","arxiv_id":"2011.09113","repositories_listed":0,"syntology":null},{"url":"/paper/privileged-knowledge-distillation-for-online","slug":"privileged-knowledge-distillation-for-online","title":"Privileged Knowledge Distillation for Online Action Detection","date":"2020-11-18","arxiv_id":"2011.09158","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-serial-number-computational-watermarking","title":"Deep Serial Number: Computational Watermarking for DNN Intellectual Property Protection","date":"2020-11-17","arxiv_id":"2011.08960","repositories_listed":0,"syntology":null},{"url":null,"slug":"digging-deeper-into-crnn-model-in-chinese","title":"Digging Deeper into CRNN Model in Chinese Text Images Recognition","date":"2020-11-17","arxiv_id":"2011.08505","repositories_listed":0,"syntology":null},{"url":null,"slug":"generalized-continual-zero-shot-learning","title":"Generalized Continual Zero-Shot Learning","date":"2020-11-17","arxiv_id":"2011.08508","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-estimating-the-training-cost-of","title":"On Estimating the Training Cost of Conversational Recommendation Systems","date":"2020-11-10","arxiv_id":"2011.05302","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensembled-ctr-prediction-via-knowledge","title":"Ensemble Knowledge Distillation for CTR Prediction","date":"2020-11-08","arxiv_id":"2011.04106","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-like-active-learning-machines","title":"Human-Like Active Learning: Machines Simulating the Human Learning Process","date":"2020-11-07","arxiv_id":"2011.03733","repositories_listed":0,"syntology":null},{"url":null,"slug":"channel-planting-for-deep-neural-networks","title":"Channel Planting for Deep Neural Networks using Knowledge Distillation","date":"2020-11-04","arxiv_id":"2011.02390","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-self-distilling-graph-neural-network","title":"On Self-Distilling Graph Neural Network","date":"2020-11-04","arxiv_id":"2011.02255","repositories_listed":0,"syntology":null},{"url":null,"slug":"paralinguistic-privacy-protection-at-the-edge","title":"Paralinguistic Privacy Protection at the Edge","date":"2020-11-04","arxiv_id":"2011.02930","repositories_listed":0,"syntology":null},{"url":null,"slug":"perceptually-guided-end-to-end-text-to-speech","title":"Learning to Maximize Speech Quality Directly Using MOS Prediction for Neural Text-to-Speech","date":"2020-11-02","arxiv_id":"2011.01174","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-prior-and-posterior","title":"Bridging the Gap between Prior and Posterior Knowledge Selection for Knowledge-Grounded Dialogue Generation","date":"2020-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"8878b35c66bbb4b0a9739fbcbdd46a8aeaed03e280996ce5b64d3fc61dcea940","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}