{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/knowledge-distillation/papers/36","list_of":"/task/knowledge-distillation","task":"Knowledge Distillation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":36,"pages_in_order":43,"rows_per_page":100,"rows":[3501,3600],"of":4240,"counts":{"archive_papers_tagged":4240,"with_a_code_link":1740,"where_syntology_ran_a_sample":451,"not_listed_spam_title":0,"listed":4240,"listed_where_code_ran":451,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":380,"every_run_a_failure_of_syntologys_instrument":71,"listed_with_a_run_with_no_instrument_failure":380,"listed_every_run_a_failure_of_syntologys_instrument":71,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/knowledge-distillation","prev":"/task/knowledge-distillation/papers/35","next":"/task/knowledge-distillation/papers/37","papers":[{"url":null,"slug":"handling-long-tailed-feature-distribution-in","title":"Handling Long-tailed Feature Distribution in AdderNets","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"unsupervised-representation-transfer-for","title":"Unsupervised Representation Transfer for Small Networks: I Believe I Can Distill On-the-Fly","date":"2021-12-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"using-a-gan-to-generate-adversarial-examples","title":"Using a GAN to Generate Adversarial Examples to Facial Image Recognition","date":"2021-11-30","arxiv_id":"2111.15213","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-federated-learning-for-aiot","title":"Efficient Federated Learning for AIoT Applications Using Knowledge Distillation","date":"2021-11-29","arxiv_id":"2111.14347","repositories_listed":0,"syntology":null},{"url":null,"slug":"improved-knowledge-distillation-via","title":"Improved Knowledge Distillation via Adversarial Collaboration","date":"2021-11-29","arxiv_id":"2111.14356","repositories_listed":0,"syntology":null},{"url":null,"slug":"egfn-efficient-geometry-feature-network-for","title":"ESGN: Efficient Stereo Geometry Network for Fast 3D Object Detection","date":"2021-11-28","arxiv_id":"2111.14055","repositories_listed":0,"syntology":null},{"url":null,"slug":"ensembling-of-distilled-models-from-multi","title":"Ensembling of Distilled Models from Multi-task Teachers for Constrained Resource Language Pairs","date":"2021-11-26","arxiv_id":"2111.13284","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-agnostic-clustering-with-self","title":"Domain-Agnostic Clustering with Self-Distillation","date":"2021-11-23","arxiv_id":"2111.12170","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrast-reconstruction-representation","title":"Contrast-reconstruction Representation Learning for Self-supervised Skeleton-based Action Recognition","date":"2021-11-22","arxiv_id":"2111.11051","repositories_listed":0,"syntology":null},{"url":null,"slug":"hierarchical-knowledge-distillation-for","title":"Hierarchical Knowledge Distillation for Dialogue Sequence Labeling","date":"2021-11-22","arxiv_id":"2111.10957","repositories_listed":0,"syntology":null},{"url":null,"slug":"local-selective-feature-distillation-for","title":"Local-Selective Feature Distillation for Single Image Super-Resolution","date":"2021-11-22","arxiv_id":"2111.10988","repositories_listed":0,"syntology":null},{"url":null,"slug":"teacher-student-training-and-triplet-loss-to","title":"Teacher-Student Training and Triplet Loss to Reduce the Effect of Drastic Face Occlusion","date":"2021-11-20","arxiv_id":"2111.10561","repositories_listed":0,"syntology":null},{"url":null,"slug":"toxicity-detection-can-be-sensitive-to-the","title":"Toxicity Detection can be Sensitive to the Conversational Context","date":"2021-11-19","arxiv_id":"2111.10223","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamically-pruning-segformer-for-efficient","title":"Dynamically pruning segformer for efficient semantic segmentation","date":"2021-11-18","arxiv_id":"2111.09499","repositories_listed":0,"syntology":null},{"url":null,"slug":"long-tailed-multi-label-retinal-diseases","title":"Hierarchical Knowledge Guided Learning for Real-world Retinal Diseases Recognition","date":"2021-11-17","arxiv_id":"2111.08913","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-flexible-multi-task-model-for-bert-serving-1","title":"A Flexible Multi-Task Model for BERT Serving","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"aligned-weight-regularizers-for-pruning","title":"Aligned Weight Regularizers for Pruning Pretrained Neural Networks","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"compositional-data-augmentation-for","title":"Compositional Data Augmentation for Abstractive Conversation Summarization","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-to-bottom-weights-decay-a-systemic","title":"Deep-to-bottom Weights Decay: A Systemic Knowledge Review Learning Technique for Transformer Layers in Knowledge Distillation","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":"/paper/enabling-multimodal-generation-on-clip-via","slug":"enabling-multimodal-generation-on-clip-via","title":"Enabling Multimodal Generation on CLIP via Vision-Language Knowledge Distillation","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-structure-distillation-for-bert","title":"Feature Structure Distillation for BERT Transferring","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-to-teach-with-student-feedback-1","title":"Learning to Teach with Student Feedback","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"making-small-language-models-better-few-shot","title":"Making Small Language Models Better Few-Shot Learners","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-granularity-contrastive-knowledge","title":"Multi-Granularity Contrastive Knowledge Distillation for Multimodal Named Entity Recognition","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-stage-distillation-framework-for-cross","title":"Multi-stage Distillation Framework for Cross-Lingual Semantic Similarity Matching","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"nvidia-nemo-neural-machine-translation","title":"NVIDIA NeMo Neural Machine Translation Systems for English-German and English-Russian News and Biomedical Tasks at WMT21","date":"2021-11-16","arxiv_id":"2111.08634","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-general-teacher-for-multi-data-multi-task","title":"One General Teacher for Multi-Data Multi-Task: A New Knowledge Distillation Framework for Discourse Relation Analysis","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"redistributing-low-frequency-words-making-the","title":"Redistributing Low-Frequency Words: Making the Most of Monolingual Data in Non-Autoregressive Translation","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"self-distilled-pruning-of-neural-networks-1","title":"Self-Distilled Pruning of Neural Networks","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-progressive-distillation-resolving-1","title":"Sparse Progressive Distillation: Resolving Overfitting under Pretrain-and-Finetune Paradigm","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"when-chosen-wisely-more-data-is-what-you-need","title":"When Chosen Wisely, More Data Is What You Need: A Universal Sample-Efficient Strategy For Data Augmentation","date":"2021-11-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"synthetic-unknown-class-learning-for-learning","title":"Synthetic Unknown Class Learning for Learning Unknowns","date":"2021-11-15","arxiv_id":"2111.08062","repositories_listed":0,"syntology":null},{"url":null,"slug":"domain-generalization-on-efficient-acoustic","title":"Domain Generalization on Efficient Acoustic Scene Classification using Residual Normalization","date":"2021-11-12","arxiv_id":"2111.06531","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-interpretation-with-explainable","title":"Learning Interpretation with Explainable Knowledge Distillation","date":"2021-11-12","arxiv_id":"2111.06945","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-on-green-deep-learning","title":"A Survey on Green Deep Learning","date":"2021-11-08","arxiv_id":"2111.05193","repositories_listed":0,"syntology":null},{"url":null,"slug":"class-token-and-knowledge-distillation-for","title":"Class Token and Knowledge Distillation for Multi-head Self-Attention Speaker Verification Systems","date":"2021-11-06","arxiv_id":"2111.03842","repositories_listed":0,"syntology":null},{"url":null,"slug":"autokd-automatic-knowledge-distillation-into","title":"AUTOKD: Automatic Knowledge Distillation Into A Student Architecture Family","date":"2021-11-05","arxiv_id":"2111.03555","repositories_listed":0,"syntology":null},{"url":null,"slug":"dvfl-a-vertical-federated-learning-method-for","title":"DVFL: A Vertical Federated Learning Method for Dynamic Data","date":"2021-11-05","arxiv_id":"2111.03341","repositories_listed":0,"syntology":null},{"url":null,"slug":"fighting-covid-19-in-the-dark-methodology-for","title":"A methodology for training homomorphicencryption friendly neural networks","date":"2021-11-05","arxiv_id":"2111.03362","repositories_listed":0,"syntology":null},{"url":null,"slug":"oracle-teacher-towards-better-knowledge","title":"Oracle Teacher: Leveraging Target Information for Better Knowledge Distillation of CTC Models","date":"2021-11-05","arxiv_id":"2111.03664","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualizing-the-emergence-of-intermediate","title":"Visualizing the Emergence of Intermediate Visual Patterns in DNNs","date":"2021-11-05","arxiv_id":"2111.03505","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-advantages-of-interactive-and-non","title":"Leveraging Advantages of Interactive and Non-Interactive Models for Vector-Based Cross-Lingual Information Retrieval","date":"2021-11-03","arxiv_id":"2111.01992","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-cross-distillation-for-membership","title":"Knowledge Cross-Distillation for Membership Privacy","date":"2021-11-02","arxiv_id":"2111.01363","repositories_listed":0,"syntology":null},{"url":null,"slug":"autosumm-automatic-model-creation-for-text","title":"AUTOSUMM: Automatic Model Creation for Text Summarization","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"combining-curriculum-learning-and-knowledge","title":"Combining Curriculum Learning and Knowledge Distillation for Dialogue Generation","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"gaml-bert-improving-bert-early-exiting-by","title":"GAML-BERT: Improving BERT Early Exiting by Gradient Aligned Mutual Learning","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"how-to-select-one-among-all-an-empirical","title":"How to Select One Among All ? An Empirical Study Towards the Robustness of Knowledge Distillation in Natural Language Understanding","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hw-tscs-participation-in-the-wmt-2021-large","title":"HW-TSC’s Participation in the WMT 2021 Large-Scale Multilingual Translation Task","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"hw-tscs-participation-in-the-wmt-2021-news","title":"HW-TSC’s Participation in the WMT 2021 News Translation Shared Task","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"limitations-of-knowledge-distillation-for","title":"Limitations of Knowledge Distillation for Zero-shot Transfer Learning","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-neural-machine-translation-can-1","title":"Multilingual Neural Machine Translation: Can Linguistic Hierarchies Help?","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"mutual-learning-improves-end-to-end-speech","title":"Mutual-Learning Improves End-to-End Speech Translation","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"nvidia-nemos-neural-machine-translation","title":"NVIDIA NeMo’s Neural Machine Translation Systems for English-German and English-Russian News and Biomedical Tasks at WMT21","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"papagos-submission-for-the-wmt21-quality","title":"Papago’s Submission for the WMT21 Quality Estimation Shared Task","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pdaln-progressive-domain-adaptation-over-a","title":"PDALN: Progressive Domain Adaptation over a Pre-trained Model for Low-Resource Cross-Domain Named Entity Recognition","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"rw-kd-sample-wise-loss-terms-re-weighting-for","title":"RW-KD: Sample-wise Loss Terms Re-Weighting for Knowledge Distillation","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"students-who-study-together-learn-better-on","title":"Students Who Study Together Learn Better: On the Importance of Collective Knowledge Distillation for Domain Transfer in Fact Verification","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"tentrans-large-scale-multilingual-machine","title":"TenTrans Large-Scale Multilingual Machine Translation System for WMT21","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-lmu-munich-system-for-the-wmt-2021-large","title":"The LMU Munich System for the WMT 2021 Large-Scale Multilingual Machine Translation Shared Task","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-mininglamp-machine-translation-system-for","title":"The Mininglamp Machine Translation System for WMT21","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"the-niutrans-system-for-the-wmt-2021","title":"The NiuTrans System for the WMT 2021 Efficiency Task","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"universal-kd-attention-based-output-grounded","title":"Universal-KD: Attention-based Output-Grounded Intermediate Layer Knowledge Distillation","date":"2021-11-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-the-knowledge-distillation-from","title":"Rethinking the Knowledge Distillation From the Perspective of Model Calibration","date":"2021-10-31","arxiv_id":"2111.01684","repositories_listed":0,"syntology":null},{"url":null,"slug":"estimating-and-maximizing-mutual-information","title":"Estimating and Maximizing Mutual Information for Knowledge Distillation","date":"2021-10-29","arxiv_id":"2110.15946","repositories_listed":0,"syntology":null},{"url":null,"slug":"model-fusion-of-heterogeneous-neural-networks-1","title":"On Cross-Layer Alignment for Model Fusion of Heterogeneous Neural Networks","date":"2021-10-29","arxiv_id":"2110.15538","repositories_listed":0,"syntology":null},{"url":null,"slug":"nxmtransformer-semi-structured-sparsification","title":"NxMTransformer: Semi-Structured Sparsification for Natural Language Understanding via ADMM","date":"2021-10-28","arxiv_id":"2110.15766","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-model-agnostic-federated-learning-1","title":"Towards Model Agnostic Federated Learning Using Knowledge Distillation","date":"2021-10-28","arxiv_id":"2110.15210","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-classification-knowledge-distillation","title":"Beyond Classification: Knowledge Distillation using Multi-Object Impressions","date":"2021-10-27","arxiv_id":"2110.14215","repositories_listed":0,"syntology":null},{"url":"/paper/temporal-knowledge-distillation-for-on-device","slug":"temporal-knowledge-distillation-for-on-device","title":"Temporal Knowledge Distillation for On-device Audio Classification","date":"2021-10-27","arxiv_id":"2110.14131","repositories_listed":0,"syntology":null},{"url":null,"slug":"response-based-distillation-for-incremental-1","title":"Response-based Distillation for Incremental Object Detection","date":"2021-10-26","arxiv_id":"2110.13471","repositories_listed":0,"syntology":null},{"url":null,"slug":"muse-feature-self-distillation-with-mutual","title":"MUSE: Feature Self-Distillation with Mutual Information and Self-Information","date":"2021-10-25","arxiv_id":"2110.12606","repositories_listed":0,"syntology":null},{"url":null,"slug":"network-compression-and-faster-inference","title":"Reconstructing Pruned Filters using Cheap Spatial Transformations","date":"2021-10-25","arxiv_id":"2110.12844","repositories_listed":0,"syntology":null},{"url":"/paper/x-distill-improving-self-supervised-monocular","slug":"x-distill-improving-self-supervised-monocular","title":"X-Distill: Improving Self-Supervised Monocular Depth via Cross-Task Distillation","date":"2021-10-24","arxiv_id":"2110.12516","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-and-when-adversarial-robustness-transfers-1","title":"How and When Adversarial Robustness Transfers in Knowledge Distillation?","date":"2021-10-22","arxiv_id":"2110.12072","repositories_listed":0,"syntology":null},{"url":null,"slug":"pseudo-supervised-monocular-depth-estimation","title":"Pseudo Supervised Monocular Depth Estimation with Teacher-Student Network","date":"2021-10-22","arxiv_id":"2110.11545","repositories_listed":0,"syntology":null},{"url":null,"slug":"augmenting-knowledge-distillation-with-peer","title":"Augmenting Knowledge Distillation With Peer-To-Peer Mutual Learning For Model Compression","date":"2021-10-21","arxiv_id":"2110.11023","repositories_listed":0,"syntology":null},{"url":null,"slug":"class-incremental-online-streaming-learning","title":"Class Incremental Online Streaming Learning","date":"2021-10-20","arxiv_id":"2110.10741","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-from-language-model-to","title":"Knowledge distillation from language model to acoustic model: a hierarchical multi-task learning approach","date":"2021-10-20","arxiv_id":"2110.10429","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-short-study-on-compressing-decoder-based","title":"A Short Study on Compressing Decoder-Based Language Models","date":"2021-10-16","arxiv_id":"2110.08460","repositories_listed":0,"syntology":null},{"url":null,"slug":"know-your-tools-well-better-textit-and-faster","title":"Know your tools well: Better $\\textit{and}$ faster QA with synthetic examples","date":"2021-10-16","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"pro-kd-progressive-distillation-by-following","title":"Pro-KD: Progressive Distillation by Following the Footsteps of the Teacher","date":"2021-10-16","arxiv_id":"2110.08532","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-do-compressed-large-language-models","title":"Robustness Challenges in Model Distillation and Pruning for Natural Language Understanding","date":"2021-10-16","arxiv_id":"2110.08419","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-multimodal-to-unimodal-attention-in","title":"From Multimodal to Unimodal Attention in Transformers using Knowledge Distillation","date":"2021-10-15","arxiv_id":"2110.08270","repositories_listed":0,"syntology":null},{"url":null,"slug":"kronecker-decomposition-for-gpt-compression","title":"Kronecker Decomposition for GPT Compression","date":"2021-10-15","arxiv_id":"2110.08152","repositories_listed":0,"syntology":null},{"url":null,"slug":"multilingual-neural-machine-translation-can","title":"Multilingual Neural Machine Translation:Can Linguistic Hierarchies Help?","date":"2021-10-15","arxiv_id":"2110.07816","repositories_listed":0,"syntology":null},{"url":null,"slug":"sparse-progressive-distillation-resolving","title":"Sparse Progressive Distillation: Resolving Overfitting under Pretrain-and-Finetune Paradigm","date":"2021-10-15","arxiv_id":"2110.08190","repositories_listed":0,"syntology":null},{"url":null,"slug":"false-negative-distillation-and-contrastive","title":"False Negative Distillation and Contrastive Learning for Personalized Outfit Recommendation","date":"2021-10-13","arxiv_id":"2110.06483","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-modelling-via-learning-to-rank","title":"Language Modelling via Learning to Rank","date":"2021-10-13","arxiv_id":"2110.06961","repositories_listed":0,"syntology":null},{"url":null,"slug":"compact-cnn-models-for-on-device-ocular-based","title":"Compact CNN Models for On-device Ocular-based User Recognition in Mobile Devices","date":"2021-10-11","arxiv_id":"2110.04953","repositories_listed":0,"syntology":null},{"url":"/paper/rectifying-the-data-bias-in-knowledge","slug":"rectifying-the-data-bias-in-knowledge","title":"Rectifying the Data Bias in Knowledge Distillation","date":"2021-10-11","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-streaming-egocentric-action","title":"Towards Streaming Egocentric Action Anticipation","date":"2021-10-11","arxiv_id":"2110.05386","repositories_listed":0,"syntology":null},{"url":null,"slug":"visualizing-the-embedding-space-to-explain","title":"Visualizing the embedding space to explain the effect of knowledge distillation","date":"2021-10-09","arxiv_id":"2110.04483","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-for-neural-transducers","title":"Knowledge Distillation for Neural Transducers from Large Self-Supervised Pre-trained Models","date":"2021-10-07","arxiv_id":"2110.03334","repositories_listed":0,"syntology":null},{"url":null,"slug":"peer-collaborative-learning-for-polyphonic","title":"Peer Collaborative Learning for Polyphonic Sound Event Detection","date":"2021-10-07","arxiv_id":"2110.03511","repositories_listed":0,"syntology":null},{"url":null,"slug":"online-hyperparameter-meta-learning-with","title":"Online Hyperparameter Meta-Learning with Hypergradient Distillation","date":"2021-10-06","arxiv_id":"2110.02508","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-interplay-between-sparsity-naturalness","title":"On the Interplay Between Sparsity, Naturalness, Intelligibility, and Prosody in Speech Synthesis","date":"2021-10-04","arxiv_id":"2110.01147","repositories_listed":0,"syntology":null},{"url":null,"slug":"student-helping-teacher-teacher-evolution-via","title":"Student Helping Teacher: Teacher Evolution via Self-Knowledge Distillation","date":"2021-10-01","arxiv_id":"2110.00329","repositories_listed":0,"syntology":null},{"url":null,"slug":"born-again-neural-rankers","title":"Improving Neural Ranking via Lossless Knowledge Distillation","date":"2021-09-30","arxiv_id":"2109.15285","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-neural-compression-via-concurrent","title":"Deep Neural Compression Via Concurrent Pruning and Self-Distillation","date":"2021-09-30","arxiv_id":"2109.15014","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-overhaul-of-distilling","title":"A Comprehensive Overhaul of Distilling Unconditional GANs","date":"2021-09-29","arxiv_id":null,"repositories_listed":0,"syntology":null}],"record_sha256":"fd5f909e3c72397066154f783d145eb7f629c44abd18afa55bb8aabe829b79c4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}