{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/knowledge-distillation/papers/23","list_of":"/method/knowledge-distillation","method":"Knowledge Distillation","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":23,"pages_in_order":31,"rows_per_page":100,"rows":[2201,2300],"of":3071,"counts":{"archive_papers_tagged":3071,"with_a_code_link":1258,"where_syntology_ran_a_sample":320,"not_listed_spam_title":0,"listed":3071,"listed_where_code_ran":320,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":276,"every_run_a_failure_of_syntologys_instrument":44,"listed_with_a_run_with_no_instrument_failure":276,"listed_every_run_a_failure_of_syntologys_instrument":44,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/knowledge-distillation","prev":"/method/knowledge-distillation/papers/22","next":"/method/knowledge-distillation/papers/24","papers":[{"paper":"/paper/role-of-data-augmentation-strategies-in","slug":"role-of-data-augmentation-strategies-in","title":"Role of Data Augmentation Strategies in Knowledge Distillation for Wearable Sensor Data","date":"2022-01-01","arxiv_id":"2201.00111","n_code_links":1,"syntology":null},{"paper":null,"slug":"conditional-generative-data-free-knowledge","title":"Conditional Generative Data-free Knowledge Distillation","date":"2021-12-31","arxiv_id":"2112.15358","n_code_links":0,"syntology":null},{"paper":null,"slug":"data-free-knowledge-transfer-a-survey","title":"Data-Free Knowledge Transfer: A Survey","date":"2021-12-31","arxiv_id":"2112.15278","n_code_links":0,"syntology":null},{"paper":null,"slug":"automatic-mixed-precision-quantization-search","title":"Automatic Mixed-Precision Quantization Search of BERT","date":"2021-12-30","arxiv_id":"2112.14938","n_code_links":0,"syntology":null},{"paper":"/paper/confidence-aware-multi-teacher-knowledge","slug":"confidence-aware-multi-teacher-knowledge","title":"Confidence-Aware Multi-Teacher Knowledge Distillation","date":"2021-12-30","arxiv_id":"2201.00007","n_code_links":1,"syntology":null},{"paper":"/paper/online-adversarial-distillation-for-graph","slug":"online-adversarial-distillation-for-graph","title":"Online Adversarial Knowledge Distillation for Graph Neural Networks","date":"2021-12-28","arxiv_id":"2112.13966","n_code_links":1,"syntology":null},{"paper":null,"slug":"adaptive-beam-search-to-enhance-on-device","title":"Adaptive Beam Search to Enhance On-device Abstractive Summarization","date":"2021-12-22","arxiv_id":"2201.02739","n_code_links":0,"syntology":null},{"paper":null,"slug":"self-distillation-mixup-training-for-non","title":"Self-Distillation Mixup Training for Non-autoregressive Neural Machine Translation","date":"2021-12-22","arxiv_id":"2112.11640","n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-modality-distillation-via-learning-the","title":"Multi-Modality Distillation via Learning the teacher's modality-level Gram Matrix","date":"2021-12-21","arxiv_id":"2112.11447","n_code_links":0,"syntology":null},{"paper":null,"slug":"supervised-graph-contrastive-pretraining-for","title":"Supervised Graph Contrastive Pretraining for Text Classification","date":"2021-12-21","arxiv_id":"2112.11389","n_code_links":0,"syntology":null},{"paper":null,"slug":"controlling-the-quality-of-distillation-in","title":"Controlling the Quality of Distillation in Response-Based Network Compression","date":"2021-12-19","arxiv_id":"2112.10047","n_code_links":0,"syntology":null},{"paper":null,"slug":"legodnn-block-grained-scaling-of-deep-neural","title":"LegoDNN: Block-grained Scaling of Deep Neural Networks for Mobile Vision","date":"2021-12-18","arxiv_id":"2112.09852","n_code_links":0,"syntology":null},{"paper":null,"slug":"distillation-of-human-object-interaction","title":"Distillation of Human-Object Interaction Contexts for Action Recognition","date":"2021-12-17","arxiv_id":"2112.09448","n_code_links":0,"syntology":null},{"paper":"/paper/pixel-distillation-a-new-knowledge","slug":"pixel-distillation-a-new-knowledge","title":"Pixel Distillation: A New Knowledge Distillation Scheme for Low-Resolution Image Recognition","date":"2021-12-17","arxiv_id":"2112.09532","n_code_links":1,"syntology":null},{"paper":"/paper/towards-disturbance-free-visual-mobile","slug":"towards-disturbance-free-visual-mobile","title":"Towards Disturbance-Free Visual Mobile Manipulation","date":"2021-12-17","arxiv_id":"2112.12612","n_code_links":1,"syntology":null},{"paper":null,"slug":"weakly-supervised-semantic-segmentation-via","title":"Weakly Supervised Semantic Segmentation via Alternative Self-Dual Teaching","date":"2021-12-17","arxiv_id":"2112.09459","n_code_links":0,"syntology":null},{"paper":"/paper/learning-cross-lingual-ir-from-an-english","slug":"learning-cross-lingual-ir-from-an-english","title":"Learning Cross-Lingual IR from an English Retriever","date":"2021-12-15","arxiv_id":"2112.08185","n_code_links":1,"syntology":null},{"paper":"/paper/a-deep-knowledge-distillation-framework-for","slug":"a-deep-knowledge-distillation-framework-for","title":"A Deep Knowledge Distillation framework for EEG assisted enhancement of single-lead ECG based sleep staging","date":"2021-12-14","arxiv_id":"2112.07252","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-a-unified-foundation-model-jointly","title":"Towards a Unified Foundation Model: Jointly Pre-Training Transformers on Unpaired Images and Text","date":"2021-12-14","arxiv_id":"2112.07074","n_code_links":0,"syntology":null},{"paper":"/paper/up-to-100x-faster-data-free-knowledge","slug":"up-to-100x-faster-data-free-knowledge","title":"Up to 100$\\times$ Faster Data-free Knowledge Distillation","date":"2021-12-12","arxiv_id":"2112.06253","n_code_links":2,"syntology":null},{"paper":"/paper/disco-effective-knowledge-distillation-for","slug":"disco-effective-knowledge-distillation-for","title":"DistilCSE: Effective Knowledge Distillation For Contrastive Sentence Embeddings","date":"2021-12-10","arxiv_id":"2112.05638","n_code_links":1,"syntology":null},{"paper":"/paper/human-interpretation-and-exploitation-of-self","slug":"human-interpretation-and-exploitation-of-self","title":"Human Guided Exploitation of Interpretable Attention Patterns in Summarization and Topic Segmentation","date":"2021-12-10","arxiv_id":"2112.05364","n_code_links":1,"syntology":null},{"paper":"/paper/mask-invariant-face-recognition-through","slug":"mask-invariant-face-recognition-through","title":"Mask-invariant Face Recognition through Template-level Knowledge Distillation","date":"2021-12-10","arxiv_id":"2112.05646","n_code_links":1,"syntology":null},{"paper":null,"slug":"boosting-contrastive-learning-with-relation","title":"Boosting Contrastive Learning with Relation Knowledge Distillation","date":"2021-12-08","arxiv_id":"2112.04174","n_code_links":0,"syntology":null},{"paper":"/paper/a-contrastive-distillation-approach-for","slug":"a-contrastive-distillation-approach-for","title":"A Contrastive Distillation Approach for Incremental Semantic Segmentation in Aerial Images","date":"2021-12-07","arxiv_id":"2112.03814","n_code_links":1,"syntology":null},{"paper":"/paper/add-frequency-attention-and-multi-view-based","slug":"add-frequency-attention-and-multi-view-based","title":"ADD: Frequency Attention and Multi-View based Knowledge Distillation to Detect Low-Quality Compressed Deepfake Images","date":"2021-12-07","arxiv_id":"2112.03553","n_code_links":2,"syntology":null},{"paper":"/paper/improving-neural-cross-lingual-summarization","slug":"improving-neural-cross-lingual-summarization","title":"Improving Neural Cross-Lingual Summarization via Employing Optimal Transport Distance for Knowledge Distillation","date":"2021-12-07","arxiv_id":"2112.03473","n_code_links":1,"syntology":null},{"paper":"/paper/classic-continual-and-contrastive-learning-of-1","slug":"classic-continual-and-contrastive-learning-of-1","title":"CLASSIC: Continual and Contrastive Learning of Aspect Sentiment Classification Tasks","date":"2021-12-05","arxiv_id":"2112.02714","n_code_links":1,"syntology":null},{"paper":null,"slug":"extracting-knowledge-from-features-with","title":"Extracting knowledge from features with multilevel abstraction","date":"2021-12-04","arxiv_id":"2112.13642","n_code_links":0,"syntology":null},{"paper":null,"slug":"kdctime-knowledge-distillation-with","title":"KDCTime: Knowledge Distillation with Calibration on InceptionTime for Time-series Classification","date":"2021-12-04","arxiv_id":"2112.02291","n_code_links":0,"syntology":null},{"paper":"/paper/a-fast-knowledge-distillation-framework-for","slug":"a-fast-knowledge-distillation-framework-for","title":"A Fast Knowledge Distillation Framework for Visual Recognition","date":"2021-12-02","arxiv_id":"2112.01528","n_code_links":2,"syntology":null},{"paper":null,"slug":"fedrad-federated-robust-adaptive-distillation","title":"FedRAD: Federated Robust Adaptive Distillation","date":"2021-12-02","arxiv_id":"2112.01405","n_code_links":0,"syntology":null},{"paper":"/paper/tiny-newsrec-efficient-and-effective-plm","slug":"tiny-newsrec-efficient-and-effective-plm","title":"Tiny-NewsRec: Effective and Efficient PLM-based News Recommendation","date":"2021-12-02","arxiv_id":"2112.00944","n_code_links":1,"syntology":null},{"paper":"/paper/distilling-meta-knowledge-on-heterogeneous","slug":"distilling-meta-knowledge-on-heterogeneous","title":"Distilling Meta Knowledge on Heterogeneous Graph for Illicit Drug Trafficker Detection on Social Media","date":"2021-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/extrapolating-from-a-single-image-to-a","slug":"extrapolating-from-a-single-image-to-a","title":"The Augmented Image Prior: Distilling 1000 Classes by Extrapolating from a Single Image","date":"2021-12-01","arxiv_id":"2112.00725","n_code_links":1,"syntology":null},{"paper":"/paper/information-theoretic-representation","slug":"information-theoretic-representation","title":"Information Theoretic Representation Distillation","date":"2021-12-01","arxiv_id":"2112.00459","n_code_links":1,"syntology":null},{"paper":"/paper/shapeshifter-a-parameter-efficient","slug":"shapeshifter-a-parameter-efficient","title":"Shapeshifter: a Parameter-efficient Transformer using Factorized Reshaped Matrices","date":"2021-12-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"unsupervised-representation-transfer-for","title":"Unsupervised Representation Transfer for Small Networks: I Believe I Can Distill On-the-Fly","date":"2021-12-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"using-a-gan-to-generate-adversarial-examples","title":"Using a GAN to Generate Adversarial Examples to Facial Image Recognition","date":"2021-11-30","arxiv_id":"2111.15213","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-federated-learning-for-aiot","title":"Efficient Federated Learning for AIoT Applications Using Knowledge Distillation","date":"2021-11-29","arxiv_id":"2111.14347","n_code_links":0,"syntology":null},{"paper":null,"slug":"improved-knowledge-distillation-via","title":"Improved Knowledge Distillation via Adversarial Collaboration","date":"2021-11-29","arxiv_id":"2111.14356","n_code_links":0,"syntology":null},{"paper":null,"slug":"egfn-efficient-geometry-feature-network-for","title":"ESGN: Efficient Stereo Geometry Network for Fast 3D Object Detection","date":"2021-11-28","arxiv_id":"2111.14055","n_code_links":0,"syntology":null},{"paper":null,"slug":"ensembling-of-distilled-models-from-multi","title":"Ensembling of Distilled Models from Multi-task Teachers for Constrained Resource Language Pairs","date":"2021-11-26","arxiv_id":"2111.13284","n_code_links":0,"syntology":null},{"paper":"/paper/wifi-based-multi-task-sensing","slug":"wifi-based-multi-task-sensing","title":"WiFi-based Multi-task Sensing","date":"2021-11-26","arxiv_id":"2111.14619","n_code_links":1,"syntology":null},{"paper":"/paper/evdistill-asynchronous-events-to-end-task-1","slug":"evdistill-asynchronous-events-to-end-task-1","title":"EvDistill: Asynchronous Events to End-task Learning via Bidirectional Reconstruction-guided Cross-modal Knowledge Distillation","date":"2021-11-24","arxiv_id":"2111.12341","n_code_links":1,"syntology":{"ran":10,"of":17,"n_ran_checked":6,"n_instrument":4,"unverified":7,"pointer_only":17,"phrase":"10 ran (of which 6 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","official":{"repos":["addisonwang2013/evdistill"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":6,"n_ran_no_instrument_failure":6,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-slimmed-vision-transformer","slug":"self-slimmed-vision-transformer","title":"Self-slimmed Vision Transformer","date":"2021-11-24","arxiv_id":"2111.12624","n_code_links":1,"syntology":null},{"paper":null,"slug":"domain-agnostic-clustering-with-self","title":"Domain-Agnostic Clustering with Self-Distillation","date":"2021-11-23","arxiv_id":"2111.12170","n_code_links":0,"syntology":null},{"paper":"/paper/focal-and-global-knowledge-distillation-for","slug":"focal-and-global-knowledge-distillation-for","title":"Focal and Global Knowledge Distillation for Detectors","date":"2021-11-23","arxiv_id":"2111.11837","n_code_links":1,"syntology":null},{"paper":"/paper/semi-online-knowledge-distillation","slug":"semi-online-knowledge-distillation","title":"Semi-Online Knowledge Distillation","date":"2021-11-23","arxiv_id":"2111.11747","n_code_links":1,"syntology":null},{"paper":null,"slug":"hierarchical-knowledge-distillation-for","title":"Hierarchical Knowledge Distillation for Dialogue Sequence Labeling","date":"2021-11-22","arxiv_id":"2111.10957","n_code_links":0,"syntology":null},{"paper":null,"slug":"local-selective-feature-distillation-for","title":"Local-Selective Feature Distillation for Single Image Super-Resolution","date":"2021-11-22","arxiv_id":"2111.10988","n_code_links":0,"syntology":null},{"paper":null,"slug":"teacher-student-training-and-triplet-loss-to","title":"Teacher-Student Training and Triplet Loss to Reduce the Effect of Drastic Face Occlusion","date":"2021-11-20","arxiv_id":"2111.10561","n_code_links":0,"syntology":null},{"paper":null,"slug":"toxicity-detection-can-be-sensitive-to-the","title":"Toxicity Detection can be Sensitive to the Conversational Context","date":"2021-11-19","arxiv_id":"2111.10223","n_code_links":0,"syntology":null},{"paper":null,"slug":"dynamically-pruning-segformer-for-efficient","title":"Dynamically pruning segformer for efficient semantic segmentation","date":"2021-11-18","arxiv_id":"2111.09499","n_code_links":0,"syntology":null},{"paper":null,"slug":"long-tailed-multi-label-retinal-diseases","title":"Hierarchical Knowledge Guided Learning for Real-world Retinal Diseases Recognition","date":"2021-11-17","arxiv_id":"2111.08913","n_code_links":0,"syntology":null},{"paper":"/paper/an-unsupervised-multiple-task-and-multiple","slug":"an-unsupervised-multiple-task-and-multiple","title":"An Unsupervised Multiple-Task and Multiple-Teacher Model for Cross-lingual Named Entity Recognition","date":"2021-11-16","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"compositional-data-augmentation-for","title":"Compositional Data Augmentation for Abstractive Conversation Summarization","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"deep-to-bottom-weights-decay-a-systemic","title":"Deep-to-bottom Weights Decay: A Systemic Knowledge Review Learning Technique for Transformer Layers in Knowledge Distillation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/enabling-multimodal-generation-on-clip-via","slug":"enabling-multimodal-generation-on-clip-via","title":"Enabling Multimodal Generation on CLIP via Vision-Language Knowledge Distillation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-teach-with-student-feedback-1","title":"Learning to Teach with Student Feedback","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"making-small-language-models-better-few-shot","title":"Making Small Language Models Better Few-Shot Learners","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-granularity-contrastive-knowledge","title":"Multi-Granularity Contrastive Knowledge Distillation for Multimodal Named Entity Recognition","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"multi-stage-distillation-framework-for-cross","title":"Multi-stage Distillation Framework for Cross-Lingual Semantic Similarity Matching","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"nvidia-nemo-neural-machine-translation","title":"NVIDIA NeMo Neural Machine Translation Systems for English-German and English-Russian News and Biomedical Tasks at WMT21","date":"2021-11-16","arxiv_id":"2111.08634","n_code_links":0,"syntology":null},{"paper":null,"slug":"one-general-teacher-for-multi-data-multi-task","title":"One General Teacher for Multi-Data Multi-Task: A New Knowledge Distillation Framework for Discourse Relation Analysis","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"sparse-progressive-distillation-resolving-1","title":"Sparse Progressive Distillation: Resolving Overfitting under Pretrain-and-Finetune Paradigm","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"when-chosen-wisely-more-data-is-what-you-need","title":"When Chosen Wisely, More Data Is What You Need: A Universal Sample-Efficient Strategy For Data Augmentation","date":"2021-11-16","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"synthetic-unknown-class-learning-for-learning","title":"Synthetic Unknown Class Learning for Learning Unknowns","date":"2021-11-15","arxiv_id":"2111.08062","n_code_links":0,"syntology":null},{"paper":"/paper/facial-landmark-points-detection-using","slug":"facial-landmark-points-detection-using","title":"Facial Landmark Points Detection Using Knowledge Distillation-Based Neural Networks","date":"2021-11-13","arxiv_id":"2111.07047","n_code_links":1,"syntology":null},{"paper":null,"slug":"learning-interpretation-with-explainable","title":"Learning Interpretation with Explainable Knowledge Distillation","date":"2021-11-12","arxiv_id":"2111.06945","n_code_links":0,"syntology":null},{"paper":"/paper/incremental-meta-learning-via-episodic-replay","slug":"incremental-meta-learning-via-episodic-replay","title":"Incremental Meta-Learning via Episodic Replay Distillation for Few-Shot Image Recognition","date":"2021-11-09","arxiv_id":"2111.04993","n_code_links":1,"syntology":null},{"paper":null,"slug":"class-token-and-knowledge-distillation-for","title":"Class Token and Knowledge Distillation for Multi-head Self-Attention Speaker Verification Systems","date":"2021-11-06","arxiv_id":"2111.03842","n_code_links":0,"syntology":null},{"paper":null,"slug":"autokd-automatic-knowledge-distillation-into","title":"AUTOKD: Automatic Knowledge Distillation Into A Student Architecture Family","date":"2021-11-05","arxiv_id":"2111.03555","n_code_links":0,"syntology":null},{"paper":"/paper/ltd-low-temperature-distillation-for-robust","slug":"ltd-low-temperature-distillation-for-robust","title":"LTD: Low Temperature Distillation for Robust Adversarial Training","date":"2021-11-03","arxiv_id":"2111.02331","n_code_links":1,"syntology":null},{"paper":null,"slug":"knowledge-cross-distillation-for-membership","title":"Knowledge Cross-Distillation for Membership Privacy","date":"2021-11-02","arxiv_id":"2111.01363","n_code_links":0,"syntology":null},{"paper":null,"slug":"autosumm-automatic-model-creation-for-text","title":"AUTOSUMM: Automatic Model Creation for Text Summarization","date":"2021-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/collaborative-learning-of-bidirectional","slug":"collaborative-learning-of-bidirectional","title":"Collaborative Learning of Bidirectional Decoders for Unsupervised Text Style Transfer","date":"2021-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"combining-curriculum-learning-and-knowledge","title":"Combining Curriculum Learning and Knowledge Distillation for Dialogue Generation","date":"2021-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/distilling-knowledge-for-empathy-detection","slug":"distilling-knowledge-for-empathy-detection","title":"Distilling Knowledge for Empathy Detection","date":"2021-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/distilling-object-detectors-with-feature","slug":"distilling-object-detectors-with-feature","title":"Distilling Object Detectors with Feature Richness","date":"2021-11-01","arxiv_id":"2111.00674","n_code_links":1,"syntology":null},{"paper":null,"slug":"gaml-bert-improving-bert-early-exiting-by","title":"GAML-BERT: Improving BERT Early Exiting by Gradient Aligned Mutual Learning","date":"2021-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/improving-stance-detection-with-multi-dataset","slug":"improving-stance-detection-with-multi-dataset","title":"Improving Stance Detection with Multi-Dataset Learning and Knowledge Distillation","date":"2021-11-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"limitations-of-knowledge-distillation-for","title":"Limitations of Knowledge Distillation for Zero-shot Transfer Learning","date":"2021-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"multilingual-neural-machine-translation-can-1","title":"Multilingual Neural Machine Translation: Can Linguistic Hierarchies Help?","date":"2021-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"mutual-learning-improves-end-to-end-speech","title":"Mutual-Learning Improves End-to-End Speech Translation","date":"2021-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"pdaln-progressive-domain-adaptation-over-a","title":"PDALN: Progressive Domain Adaptation over a Pre-trained Model for Low-Resource Cross-Domain Named Entity Recognition","date":"2021-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/pp-shitu-a-practical-lightweight-image","slug":"pp-shitu-a-practical-lightweight-image","title":"PP-ShiTu: A Practical Lightweight Image Recognition System","date":"2021-11-01","arxiv_id":"2111.00775","n_code_links":2,"syntology":null},{"paper":null,"slug":"students-who-study-together-learn-better-on","title":"Students Who Study Together Learn Better: On the Importance of Collective Knowledge Distillation for Domain Transfer in Fact Verification","date":"2021-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"universal-kd-attention-based-output-grounded","title":"Universal-KD: Attention-based Output-Grounded Intermediate Layer Knowledge Distillation","date":"2021-11-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"estimating-and-maximizing-mutual-information","title":"Estimating and Maximizing Mutual Information for Knowledge Distillation","date":"2021-10-29","arxiv_id":"2110.15946","n_code_links":0,"syntology":null},{"paper":null,"slug":"model-fusion-of-heterogeneous-neural-networks-1","title":"On Cross-Layer Alignment for Model Fusion of Heterogeneous Neural Networks","date":"2021-10-29","arxiv_id":"2110.15538","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-model-agnostic-federated-learning-1","title":"Towards Model Agnostic Federated Learning Using Knowledge Distillation","date":"2021-10-28","arxiv_id":"2110.15210","n_code_links":0,"syntology":null},{"paper":"/paper/temporal-knowledge-distillation-for-on-device","slug":"temporal-knowledge-distillation-for-on-device","title":"Temporal Knowledge Distillation for On-device Audio Classification","date":"2021-10-27","arxiv_id":"2110.14131","n_code_links":0,"syntology":null},{"paper":null,"slug":"response-based-distillation-for-incremental-1","title":"Response-based Distillation for Incremental Object Detection","date":"2021-10-26","arxiv_id":"2110.13471","n_code_links":0,"syntology":null},{"paper":"/paper/instance-conditional-knowledge-distillation","slug":"instance-conditional-knowledge-distillation","title":"Instance-Conditional Knowledge Distillation for Object Detection","date":"2021-10-25","arxiv_id":"2110.12724","n_code_links":1,"syntology":null},{"paper":null,"slug":"muse-feature-self-distillation-with-mutual","title":"MUSE: Feature Self-Distillation with Mutual Information and Self-Information","date":"2021-10-25","arxiv_id":"2110.12606","n_code_links":0,"syntology":null},{"paper":null,"slug":"network-compression-and-faster-inference","title":"Reconstructing Pruned Filters using Cheap Spatial Transformations","date":"2021-10-25","arxiv_id":"2110.12844","n_code_links":0,"syntology":null},{"paper":"/paper/x-distill-improving-self-supervised-monocular","slug":"x-distill-improving-self-supervised-monocular","title":"X-Distill: Improving Self-Supervised Monocular Depth via Cross-Task Distillation","date":"2021-10-24","arxiv_id":"2110.12516","n_code_links":0,"syntology":null},{"paper":"/paper/pixel-by-pixel-cross-domain-alignment-for-few","slug":"pixel-by-pixel-cross-domain-alignment-for-few","title":"Pixel-by-Pixel Cross-Domain Alignment for Few-Shot Semantic Segmentation","date":"2021-10-22","arxiv_id":"2110.11650","n_code_links":1,"syntology":null},{"paper":null,"slug":"knowledge-distillation-from-language-model-to","title":"Knowledge distillation from language model to acoustic model: a hierarchical multi-task learning approach","date":"2021-10-20","arxiv_id":"2110.10429","n_code_links":0,"syntology":null}],"record_sha256":"21d2ac30997cab04e85d724572ae15c424292fa9490eb220b18d4e65f9e030df","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}