{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/knowledge-distillation/papers/24","list_of":"/method/knowledge-distillation","method":"Knowledge Distillation","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":24,"pages_in_order":31,"rows_per_page":100,"rows":[2301,2400],"of":3071,"counts":{"archive_papers_tagged":3071,"with_a_code_link":1258,"where_syntology_ran_a_sample":320,"not_listed_spam_title":0,"listed":3071,"listed_where_code_ran":320,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":276,"every_run_a_failure_of_syntologys_instrument":44,"listed_with_a_run_with_no_instrument_failure":276,"listed_every_run_a_failure_of_syntologys_instrument":44,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/knowledge-distillation","prev":"/method/knowledge-distillation/papers/23","next":"/method/knowledge-distillation/papers/25","papers":[{"paper":"/paper/adaptive-distillation-aggregating-knowledge","slug":"adaptive-distillation-aggregating-knowledge","title":"Adaptive Distillation: Aggregating Knowledge from Multiple Paths for Efficient Distillation","date":"2021-10-19","arxiv_id":"2110.09674","n_code_links":1,"syntology":null},{"paper":"/paper/graph-less-neural-networks-teaching-old-mlps-1","slug":"graph-less-neural-networks-teaching-old-mlps-1","title":"Graph-less Neural Networks: Teaching Old MLPs New Tricks via Distillation","date":"2021-10-17","arxiv_id":"2110.08727","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":6,"n_instrument":1,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["snap-research/graphless-neural-networks"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-short-study-on-compressing-decoder-based","title":"A Short Study on Compressing Decoder-Based Language Models","date":"2021-10-16","arxiv_id":"2110.08460","n_code_links":0,"syntology":null},{"paper":"/paper/hrkd-hierarchical-relational-knowledge","slug":"hrkd-hierarchical-relational-knowledge","title":"HRKD: Hierarchical Relational Knowledge Distillation for Cross-domain Language Model Compression","date":"2021-10-16","arxiv_id":"2110.08551","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","official":{"repos":["cheneydon/hrkd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"pro-kd-progressive-distillation-by-following","title":"Pro-KD: Progressive Distillation by Following the Footsteps of the Teacher","date":"2021-10-16","arxiv_id":"2110.08532","n_code_links":0,"syntology":null},{"paper":null,"slug":"what-do-compressed-large-language-models","title":"Robustness Challenges in Model Distillation and Pruning for Natural Language Understanding","date":"2021-10-16","arxiv_id":"2110.08419","n_code_links":0,"syntology":null},{"paper":null,"slug":"from-multimodal-to-unimodal-attention-in","title":"From Multimodal to Unimodal Attention in Transformers using Knowledge Distillation","date":"2021-10-15","arxiv_id":"2110.08270","n_code_links":0,"syntology":null},{"paper":null,"slug":"kronecker-decomposition-for-gpt-compression","title":"Kronecker Decomposition for GPT Compression","date":"2021-10-15","arxiv_id":"2110.08152","n_code_links":0,"syntology":null},{"paper":null,"slug":"multilingual-neural-machine-translation-can","title":"Multilingual Neural Machine Translation:Can Linguistic Hierarchies Help?","date":"2021-10-15","arxiv_id":"2110.07816","n_code_links":0,"syntology":null},{"paper":null,"slug":"sparse-progressive-distillation-resolving","title":"Sparse Progressive Distillation: Resolving Overfitting under Pretrain-and-Finetune Paradigm","date":"2021-10-15","arxiv_id":"2110.08190","n_code_links":0,"syntology":null},{"paper":"/paper/clonalnet-classifying-better-by-focusing-on","slug":"clonalnet-classifying-better-by-focusing-on","title":"FocusNet: Classifying Better by Focusing on Confusing Classes","date":"2021-10-14","arxiv_id":"2110.07307","n_code_links":2,"syntology":null},{"paper":"/paper/symbolic-knowledge-distillation-from-general","slug":"symbolic-knowledge-distillation-from-general","title":"Symbolic Knowledge Distillation: from General Language Models to Commonsense Models","date":"2021-10-14","arxiv_id":"2110.07178","n_code_links":1,"syntology":null},{"paper":null,"slug":"false-negative-distillation-and-contrastive","title":"False Negative Distillation and Contrastive Learning for Personalized Outfit Recommendation","date":"2021-10-13","arxiv_id":"2110.06483","n_code_links":0,"syntology":null},{"paper":null,"slug":"language-modelling-via-learning-to-rank","title":"Language Modelling via Learning to Rank","date":"2021-10-13","arxiv_id":"2110.06961","n_code_links":0,"syntology":null},{"paper":"/paper/object-dgcnn-3d-object-detection-using","slug":"object-dgcnn-3d-object-detection-using","title":"Object DGCNN: 3D Object Detection using Dynamic Graphs","date":"2021-10-13","arxiv_id":"2110.06923","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["wangyueft/detr3d"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"compact-cnn-models-for-on-device-ocular-based","title":"Compact CNN Models for On-device Ocular-based User Recognition in Mobile Devices","date":"2021-10-11","arxiv_id":"2110.04953","n_code_links":0,"syntology":null},{"paper":null,"slug":"towards-streaming-egocentric-action","title":"Towards Streaming Egocentric Action Anticipation","date":"2021-10-11","arxiv_id":"2110.05386","n_code_links":0,"syntology":null},{"paper":"/paper/towards-data-free-domain-generalization","slug":"towards-data-free-domain-generalization","title":"Towards Data-Free Domain Generalization","date":"2021-10-09","arxiv_id":"2110.04545","n_code_links":1,"syntology":{"ran":15,"of":20,"n_ran_checked":7,"n_instrument":8,"unverified":5,"pointer_only":0,"phrase":"15 ran (of which 5 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 0 violated, 5 with no contract checked; 8 where Syntology's instrument failed) · 5 unverified","official":{"repos":["HaokunChen245/DFDG"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":5,"n_ran_no_instrument_failure":7,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"visualizing-the-embedding-space-to-explain","title":"Visualizing the embedding space to explain the effect of knowledge distillation","date":"2021-10-09","arxiv_id":"2110.04483","n_code_links":0,"syntology":null},{"paper":"/paper/cross-modal-knowledge-distillation-for-vision","slug":"cross-modal-knowledge-distillation-for-vision","title":"Cross-modal Knowledge Distillation for Vision-to-Sensor Action Recognition","date":"2021-10-08","arxiv_id":"2112.01849","n_code_links":1,"syntology":null},{"paper":null,"slug":"knowledge-distillation-for-neural-transducers","title":"Knowledge Distillation for Neural Transducers from Large Self-Supervised Pre-trained Models","date":"2021-10-07","arxiv_id":"2110.03334","n_code_links":0,"syntology":null},{"paper":null,"slug":"peer-collaborative-learning-for-polyphonic","title":"Peer Collaborative Learning for Polyphonic Sound Event Detection","date":"2021-10-07","arxiv_id":"2110.03511","n_code_links":0,"syntology":null},{"paper":"/paper/towards-accurate-cross-domain-in-bed-human","slug":"towards-accurate-cross-domain-in-bed-human","title":"Towards Accurate Cross-Domain In-Bed Human Pose Estimation","date":"2021-10-07","arxiv_id":"2110.03578","n_code_links":1,"syntology":null},{"paper":"/paper/federated-distillation-of-natural-language","slug":"federated-distillation-of-natural-language","title":"KNOT: Knowledge Distillation using Optimal Transport for Solving NLP Tasks","date":"2021-10-06","arxiv_id":"2110.02432","n_code_links":2,"syntology":null},{"paper":"/paper/inter-domain-alignment-for-predicting-high","slug":"inter-domain-alignment-for-predicting-high","title":"Inter-Domain Alignment for Predicting High-Resolution Brain Networks Using Teacher-Student Learning","date":"2021-10-06","arxiv_id":"2110.03452","n_code_links":1,"syntology":null},{"paper":null,"slug":"on-the-interplay-between-sparsity-naturalness","title":"On the Interplay Between Sparsity, Naturalness, Intelligibility, and Prosody in Speech Synthesis","date":"2021-10-04","arxiv_id":"2110.01147","n_code_links":0,"syntology":null},{"paper":"/paper/stem-an-approach-to-multi-source-domain-1","slug":"stem-an-approach-to-multi-source-domain-1","title":"STEM: An Approach to Multi-Source Domain Adaptation With Guarantees","date":"2021-10-01","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":null,"slug":"born-again-neural-rankers","title":"Improving Neural Ranking via Lossless Knowledge Distillation","date":"2021-09-30","arxiv_id":"2109.15285","n_code_links":0,"syntology":null},{"paper":"/paper/prune-your-model-before-distill-it","slug":"prune-your-model-before-distill-it","title":"Prune Your Model Before Distill It","date":"2021-09-30","arxiv_id":"2109.14960","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-comprehensive-overhaul-of-distilling","title":"A Comprehensive Overhaul of Distilling Unconditional GANs","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"a-unified-knowledge-distillation-framework","title":"A Unified Knowledge Distillation Framework for Deep Directed Graphical Models","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"automated-channel-pruning-with-learned","title":"Automated Channel Pruning with Learned Importance","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"distilling-gans-with-style-mixed-triplets-for","title":"Distilling GANs with Style-Mixed Triplets for X2I Translation with Limited Data","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"fast-and-efficient-once-for-all-networks-for","title":"Fast and Efficient Once-For-All Networks for Diverse Hardware Deployment","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"feature-kernel-distillation","title":"Feature Kernel Distillation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"generate-annotate-and-learn-generative-models-1","title":"Generate, Annotate, and Learn: Generative Models Advance Self-Training and Knowledge Distillation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"moba-multi-teacher-model-based-reinforcement","title":"MOBA: Multi-teacher Model Based Reinforcement Learning","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"neural-architecture-search-via-ensemble-based","title":"Neural Architecture Search via Ensemble-based Knowledge Distillation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"not-all-regions-are-worthy-to-be-distilled","title":"Not All Regions are Worthy to be Distilled: Region-aware Knowledge Distillation Towards Efficient Image-to-Image Translation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"prototypical-contrastive-predictive-coding","title":"Prototypical Contrastive Predictive Coding","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"reducing-the-teacher-student-gap-via-adaptive","title":"Reducing the Teacher-Student Gap via Adaptive Temperatures","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"representation-consolidation-from-multiple","title":"Representation Consolidation from Multiple Expert Teachers","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"scale-invariant-teaching-for-semi-supervised","title":"Scale-Invariant Teaching for Semi-Supervised Object Detection","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"self-slimming-vision-transformer","title":"Self-Slimming Vision Transformer","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"self-supervised-models-are-good-teaching","title":"Self-supervised Models are Good Teaching Assistants for Vision Transformers","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"seqpate-differentially-private-text","title":"SeqPATE: Differentially Private Text Generation via Knowledge Distillation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"source-target-unified-knowledge-distillation","title":"Source-Target Unified Knowledge Distillation for Memory-Efficient Federated Domain Adaptation on Edge Devices","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"to-smooth-or-not-to-smooth-on-compatibility","title":"To Smooth or not to Smooth? On Compatibility between Label Smoothing and Knowledge Distillation","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"wakening-past-concepts-without-past-data","title":"Wakening Past Concepts without Past Data: Class-incremental Learning from Placebos","date":"2021-09-29","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":"/paper/improving-question-answering-performance","slug":"improving-question-answering-performance","title":"Improving Question Answering Performance Using Knowledge Distillation and Active Learning","date":"2021-09-26","arxiv_id":"2109.12662","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mirbostani/QA-KD-AL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"partial-to-whole-knowledge-distillation","title":"Partial to Whole Knowledge Distillation: Progressive Distilling Decomposed Knowledge Boosts Student Better","date":"2021-09-26","arxiv_id":"2109.12507","n_code_links":0,"syntology":null},{"paper":"/paper/dynamic-knowledge-distillation-for-pre","slug":"dynamic-knowledge-distillation-for-pre","title":"Dynamic Knowledge Distillation for Pre-trained Language Models","date":"2021-09-23","arxiv_id":"2109.11295","n_code_links":1,"syntology":null},{"paper":null,"slug":"k-aid-enhancing-pre-trained-language-models","title":"K-AID: Enhancing Pre-trained Language Models with Domain Knowledge for Question Answering","date":"2021-09-22","arxiv_id":"2109.10547","n_code_links":0,"syntology":null},{"paper":"/paper/kd-vlp-improving-end-to-end-vision-and","slug":"kd-vlp-improving-end-to-end-vision-and","title":"KD-VLP: Improving End-to-End Vision-and-Language Pretraining with Object Knowledge Distillation","date":"2021-09-22","arxiv_id":"2109.10504","n_code_links":1,"syntology":null},{"paper":null,"slug":"low-latency-incremental-text-to-speech","title":"Low-Latency Incremental Text-to-Speech Synthesis with Distilled Context Prediction Network","date":"2021-09-22","arxiv_id":"2109.10724","n_code_links":0,"syntology":null},{"paper":null,"slug":"rail-kd-random-intermediate-layer-mapping-for","title":"RAIL-KD: RAndom Intermediate Layer Mapping for Knowledge Distillation","date":"2021-09-21","arxiv_id":"2109.10164","n_code_links":0,"syntology":null},{"paper":"/paper/distilling-linguistic-context-for-language","slug":"distilling-linguistic-context-for-language","title":"Distilling Linguistic Context for Language Model Compression","date":"2021-09-17","arxiv_id":"2109.08359","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":7,"n_instrument":1,"unverified":3,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["geondopark/ckd"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"towards-full-utilization-on-mask-task-for","title":"Towards Full Utilization on Mask Task for Distilling PLMs into NMT","date":"2021-09-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"label-assignment-distillation-for-object","title":"Label Assignment Distillation for Object Detection","date":"2021-09-16","arxiv_id":"2109.07843","n_code_links":0,"syntology":null},{"paper":"/paper/the-niutrans-system-for-the-wmt21-efficiency","slug":"the-niutrans-system-for-the-wmt21-efficiency","title":"The NiuTrans System for the WMT21 Efficiency Task","date":"2021-09-16","arxiv_id":"2109.08003","n_code_links":1,"syntology":null},{"paper":"/paper/e-fficient-bert-progressively-searching","slug":"e-fficient-bert-progressively-searching","title":"EfficientBERT: Progressively Searching Multilayer Perceptron via Warm-up Knowledge Distillation","date":"2021-09-15","arxiv_id":"2109.07222","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"0 ran · 1 unverified","official":{"repos":["cheneydon/efficient-bert"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"new-perspective-on-progressive-gans","title":"New Perspective on Progressive GANs Distillation for One-class Novelty Detection","date":"2021-09-15","arxiv_id":"2109.07295","n_code_links":0,"syntology":null},{"paper":null,"slug":"building-accurate-simple-models-with-multihop","title":"Multihop: Leveraging Complex Models to Learn Accurate Simple Models","date":"2021-09-14","arxiv_id":"2109.06961","n_code_links":0,"syntology":null},{"paper":null,"slug":"exploring-the-connection-between-knowledge","title":"A Note on Knowledge Distillation Loss Function for Object Classification","date":"2021-09-14","arxiv_id":"2109.06458","n_code_links":0,"syntology":null},{"paper":"/paper/multi-scale-aligned-distillation-for-low-1","slug":"multi-scale-aligned-distillation-for-low-1","title":"Multi-Scale Aligned Distillation for Low-Resolution Detection","date":"2021-09-14","arxiv_id":"2109.06875","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Jia-Research-Lab/MSAD","dvlab-research/msad"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"secure-your-ride-real-time-matching-success","title":"Secure Your Ride: Real-time Matching Success Rate Prediction for Passenger-Driver Pairs","date":"2021-09-14","arxiv_id":"2109.07571","n_code_links":0,"syntology":null},{"paper":null,"slug":"kroneckerbert-learning-kronecker","title":"KroneckerBERT: Learning Kronecker Decomposition for Pre-trained Language Models via Knowledge Distillation","date":"2021-09-13","arxiv_id":"2109.06243","n_code_links":0,"syntology":null},{"paper":null,"slug":"unims-a-unified-framework-for-multimodal","title":"UniMS: A Unified Framework for Multimodal Summarization with Knowledge Distillation","date":"2021-09-13","arxiv_id":"2109.05812","n_code_links":0,"syntology":null},{"paper":null,"slug":"federated-ensemble-model-based-reinforcement","title":"Federated Ensemble Model-based Reinforcement Learning in Edge Computing","date":"2021-09-12","arxiv_id":"2109.05549","n_code_links":0,"syntology":null},{"paper":null,"slug":"on-the-efficiency-of-subclass-knowledge","title":"On the Efficiency of Subclass Knowledge Distillation in Classification Tasks","date":"2021-09-12","arxiv_id":"2109.05587","n_code_links":0,"syntology":null},{"paper":null,"slug":"learning-to-teach-with-student-feedback","title":"Learning to Teach with Student Feedback","date":"2021-09-10","arxiv_id":"2109.04641","n_code_links":0,"syntology":null},{"paper":"/paper/towards-developing-a-multilingual-and-code","slug":"towards-developing-a-multilingual-and-code","title":"Towards Developing a Multilingual and Code-Mixed Visual Question Answering System by Knowledge Distillation","date":"2021-09-10","arxiv_id":"2109.04653","n_code_links":0,"syntology":null},{"paper":"/paper/beyond-preserved-accuracy-evaluating-loyalty","slug":"beyond-preserved-accuracy-evaluating-loyalty","title":"Beyond Preserved Accuracy: Evaluating Loyalty and Robustness of BERT Compression","date":"2021-09-07","arxiv_id":"2109.03228","n_code_links":1,"syntology":null},{"paper":"/paper/knowledge-distillation-using-hierarchical","slug":"knowledge-distillation-using-hierarchical","title":"Knowledge Distillation Using Hierarchical Self-Supervision Augmented Distribution","date":"2021-09-07","arxiv_id":"2109.03075","n_code_links":1,"syntology":null},{"paper":null,"slug":"towards-learning-spatially-discriminative","title":"CAM-loss: Towards Learning Spatially Discriminative Feature Representations","date":"2021-09-03","arxiv_id":"2109.01359","n_code_links":0,"syntology":null},{"paper":"/paper/catastrophic-interference-in-reinforcement","slug":"catastrophic-interference-in-reinforcement","title":"Catastrophic Interference in Reinforcement Learning: A Solution Based on Context Division and Knowledge Distillation","date":"2021-09-01","arxiv_id":"2109.00525","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sweety-dm/interference-aware-deep-q-learning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"decoupled-transformer-for-scalable-inference-1","title":"Decoupled Transformer for Scalable Inference in Open-domain Question Answering","date":"2021-09-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"knowledge-distillation-with-bert-for-image","title":"Knowledge Distillation with BERT for Image Tag-Based Privacy Prediction","date":"2021-09-01","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"lipschitz-continuity-guided-knowledge","title":"Lipschitz Continuity Guided Knowledge Distillation","date":"2021-08-29","arxiv_id":"2108.12905","n_code_links":0,"syntology":null},{"paper":"/paper/distilling-the-knowledge-of-large-scale","slug":"distilling-the-knowledge-of-large-scale","title":"Distilling the Knowledge of Large-scale Generative Models into Retrieval Models for Efficient Open-domain Conversation","date":"2021-08-28","arxiv_id":"2108.12582","n_code_links":1,"syntology":null},{"paper":null,"slug":"coco-distillnet-a-cross-layer-correlation","title":"CoCo DistillNet: a Cross-layer Correlation Distillation Network for Pathological Gastric Cancer Segmentation","date":"2021-08-27","arxiv_id":"2108.12173","n_code_links":0,"syntology":null},{"paper":null,"slug":"efficient-training-of-lightweight-neural","title":"Efficient training of lightweight neural networks using Online Self-Acquired Knowledge Distillation","date":"2021-08-26","arxiv_id":"2108.11798","n_code_links":0,"syntology":null},{"paper":"/paper/pocketnet-extreme-lightweight-face","slug":"pocketnet-extreme-lightweight-face","title":"PocketNet: Extreme Lightweight Face Recognition Network using Neural Architecture Search and Multi-Step Knowledge Distillation","date":"2021-08-24","arxiv_id":"2108.10710","n_code_links":1,"syntology":null},{"paper":null,"slug":"deploying-a-bert-based-query-title-relevance","title":"Deploying a BERT-based Query-Title Relevance Classifier in a Production System: a View from the Trenches","date":"2021-08-23","arxiv_id":"2108.10197","n_code_links":0,"syntology":null},{"paper":"/paper/supervised-compression-for-resource","slug":"supervised-compression-for-resource","title":"Supervised Compression for Resource-Constrained Edge Computing Systems","date":"2021-08-21","arxiv_id":"2108.11898","n_code_links":2,"syntology":{"ran":11,"of":17,"n_ran_checked":10,"n_instrument":1,"unverified":6,"pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["yoshitomo-matsubara/supervised-compression"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["listed","official","unlocated"]}}},{"paper":null,"slug":"knowledge-distillation-from-ensemble-of","title":"Boosting of Head Pose Estimation by Knowledge Distillation","date":"2021-08-20","arxiv_id":"2108.09183","n_code_links":0,"syntology":null},{"paper":"/paper/revisiting-adversarial-robustness","slug":"revisiting-adversarial-robustness","title":"Revisiting Adversarial Robustness Distillation: Robust Soft Labels Make Student Better","date":"2021-08-18","arxiv_id":"2108.07969","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":2,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zibojia/rslad"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"bert-learns-to-teach-knowledge-distillation","title":"BERT Learns to Teach: Knowledge Distillation with Meta Learning","date":"2021-08-17","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"g-detkd-towards-general-distillation","title":"G-DetKD: Towards General Distillation Framework for Object Detectors via Contrastive and Semantic-guided Feature Imitation","date":"2021-08-17","arxiv_id":"2108.07482","n_code_links":0,"syntology":null},{"paper":null,"slug":"online-continual-learning-for-visual-food","title":"Online Continual Learning For Visual Food Classification","date":"2021-08-15","arxiv_id":"2108.06781","n_code_links":0,"syntology":null},{"paper":"/paper/agkd-bml-defense-against-adversarial-attack","slug":"agkd-bml-defense-against-adversarial-attack","title":"AGKD-BML: Defense Against Adversarial Attack by Attention Guided Knowledge Distillation and Bi-directional Metric Learning","date":"2021-08-13","arxiv_id":"2108.06017","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hongw579/agkd-bml"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"learning-from-matured-dumb-teacher-for-fine","title":"Learning from Matured Dumb Teacher for Fine Generalization","date":"2021-08-12","arxiv_id":"2108.05776","n_code_links":0,"syntology":null},{"paper":null,"slug":"preventing-catastrophic-forgetting-and","title":"Preventing Catastrophic Forgetting and Distribution Mismatch in Knowledge Distillation via Synthetic Data","date":"2021-08-11","arxiv_id":"2108.05698","n_code_links":0,"syntology":null},{"paper":"/paper/semi-supervised-domain-generalizable-person","slug":"semi-supervised-domain-generalizable-person","title":"Semi-Supervised Domain Generalizable Person Re-Identification","date":"2021-08-11","arxiv_id":"2108.05045","n_code_links":3,"syntology":{"ran":6,"of":7,"n_ran_checked":2,"n_instrument":4,"unverified":1,"pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["JDAI-CV/fast-reid"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"learning-an-augmented-rgb-representation-with","title":"Learning an Augmented RGB Representation with Cross-Modal Knowledge Distillation for Action Detection","date":"2021-08-08","arxiv_id":"2108.03619","n_code_links":0,"syntology":null},{"paper":null,"slug":"a-distillation-based-approach-for-the","title":"A distillation based approach for the diagnosis of diseases","date":"2021-08-07","arxiv_id":"2108.03470","n_code_links":0,"syntology":null},{"paper":null,"slug":"spatio-temporal-attention-mechanism-and","title":"Spatio-Temporal Attention Mechanism and Knowledge Distillation for Lip Reading","date":"2021-08-07","arxiv_id":"2108.03543","n_code_links":0,"syntology":null},{"paper":"/paper/transferring-knowledge-distillation-for","slug":"transferring-knowledge-distillation-for","title":"Transferring Knowledge Distillation for Multilingual Social Event Detection","date":"2021-08-06","arxiv_id":"2108.03084","n_code_links":2,"syntology":null},{"paper":null,"slug":"decoupled-transformer-for-scalable-inference","title":"Decoupled Transformer for Scalable Inference in Open-domain Question Answering","date":"2021-08-05","arxiv_id":"2108.02765","n_code_links":0,"syntology":null},{"paper":"/paper/knowledge-distillation-from-bert-transformer","slug":"knowledge-distillation-from-bert-transformer","title":"Knowledge Distillation from BERT Transformer to Speech Transformer for Intent Classification","date":"2021-08-05","arxiv_id":"2108.02598","n_code_links":1,"syntology":null}],"record_sha256":"542b05527b6cc2931f0a315df262e38ac654f731bfc4e5137e8344e49b36e9af","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}