{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/knowledge-distillation/papers/38","list_of":"/task/knowledge-distillation","task":"Knowledge Distillation","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":38,"pages_in_order":43,"rows_per_page":100,"rows":[3701,3800],"of":4240,"counts":{"archive_papers_tagged":4240,"with_a_code_link":1740,"where_syntology_ran_a_sample":451,"not_listed_spam_title":0,"listed":4240,"listed_where_code_ran":451,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":380,"every_run_a_failure_of_syntologys_instrument":71,"listed_with_a_run_with_no_instrument_failure":380,"listed_every_run_a_failure_of_syntologys_instrument":71,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/knowledge-distillation","prev":"/task/knowledge-distillation/papers/37","next":"/task/knowledge-distillation/papers/39","papers":[{"url":null,"slug":"confidence-conditioned-knowledge-distillation","title":"Confidence Conditioned Knowledge Distillation","date":"2021-07-06","arxiv_id":"2107.06993","repositories_listed":0,"syntology":null},{"url":null,"slug":"embracing-the-dark-knowledge-domain","title":"Embracing the Dark Knowledge: Domain Generalization Using Regularized Knowledge Distillation","date":"2021-07-06","arxiv_id":"2107.02629","repositories_listed":0,"syntology":null},{"url":null,"slug":"modeling-accurate-human-activity-recognition","title":"A Light-weight Deep Human Activity Recognition Algorithm Using Multi-knowledge Distillation","date":"2021-07-06","arxiv_id":"2107.07331","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-distribution-of-penultimate","title":"On The Distribution of Penultimate Activations of Classification Networks","date":"2021-07-05","arxiv_id":"2107.01900","repositories_listed":0,"syntology":null},{"url":null,"slug":"audio-oriented-multimodal-machine","title":"Audio-Oriented Multimodal Machine Comprehension: Task, Dataset and Model","date":"2021-07-04","arxiv_id":"2107.01571","repositories_listed":0,"syntology":null},{"url":null,"slug":"isotonic-data-augmentation-for-knowledge","title":"Isotonic Data Augmentation for Knowledge Distillation","date":"2021-07-03","arxiv_id":"2107.01412","repositories_listed":0,"syntology":null},{"url":null,"slug":"espnet-st-iwslt-2021-offline-speech","title":"ESPnet-ST IWSLT 2021 Offline Speech Translation System","date":"2021-07-01","arxiv_id":"2107.00636","repositories_listed":0,"syntology":null},{"url":null,"slug":"global-knowledge-distillation-in-federated","title":"Local-Global Knowledge Distillation in Heterogeneous Federated Learning with Non-IID Data","date":"2021-06-30","arxiv_id":"2107.00051","repositories_listed":0,"syntology":null},{"url":null,"slug":"reward-based-1-bit-compressed-federated","title":"Reward-Based 1-bit Compressed Federated Distillation on Blockchain","date":"2021-06-27","arxiv_id":"2106.14265","repositories_listed":0,"syntology":null},{"url":null,"slug":"adapt-and-distill-developing-small-fast-and","title":"Adapt-and-Distill: Developing Small, Fast and Effective Pretrained Language Models for Domains","date":"2021-06-25","arxiv_id":"2106.13474","repositories_listed":0,"syntology":null},{"url":null,"slug":"pqk-model-compression-via-pruning","title":"PQK: Model Compression via Pruning, Quantization, and Knowledge Distillation","date":"2021-06-25","arxiv_id":"2106.14681","repositories_listed":0,"syntology":null},{"url":null,"slug":"dealing-with-training-and-test-segmentation","title":"Dealing with training and test segmentation mismatch: FBK@IWSLT2021","date":"2021-06-23","arxiv_id":"2106.12607","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-inference-via-universal-lsh-kernel","title":"Efficient Inference via Universal LSH Kernel","date":"2021-06-21","arxiv_id":"2106.11426","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-via-instance-level","title":"Knowledge Distillation via Instance-level Sequence Learning","date":"2021-06-21","arxiv_id":"2106.10885","repositories_listed":0,"syntology":null},{"url":null,"slug":"capsulerrt-relationships-aware-regression","title":"CapsuleRRT: Relationships-Aware Regression Tracking via Capsules","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"data-free-knowledge-distillation-for-image","title":"Data-Free Knowledge Distillation for Image Super-Resolution","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"minimally-invasive-surgery-for-sparse-neural","title":"Minimally Invasive Surgery for Sparse Neural Networks in Contrastive Manner","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"positive-unlabeled-data-purification-in-the","title":"Positive-Unlabeled Data Purification in the Wild for Object Detection","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"space-time-distillation-for-video-super","title":"Space-Time Distillation for Video Super-Resolution","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"teacher-s-pet-understanding-and-mitigating","title":"Teacher's pet: understanding and mitigating biases in distillation","date":"2021-06-19","arxiv_id":"2106.10494","repositories_listed":0,"syntology":null},{"url":null,"slug":"tree-like-decision-distillation","title":"Tree-Like Decision Distillation","date":"2021-06-19","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"recurrent-stacking-of-layers-in-neural","title":"Recurrent Stacking of Layers in Neural Networks: An Application to Neural Machine Translation","date":"2021-06-18","arxiv_id":"2106.10002","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-teacher-class-incremental-learning-with","title":"Dual-Teacher Class-Incremental Learning With Data-Free Generative Replay","date":"2021-06-17","arxiv_id":"2106.09835","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynamic-knowledge-distillation-with-a-single","title":"Dynamic Knowledge Distillation With Noise Elimination for RGB-D Salient Object Detection","date":"2021-06-17","arxiv_id":"2106.09517","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-from-multi-modal-to","title":"Knowledge distillation from multi-modal to mono-modal segmentation networks","date":"2021-06-17","arxiv_id":"2106.09564","repositories_listed":0,"syntology":null},{"url":null,"slug":"topology-distillation-for-recommender-system","title":"Topology Distillation for Recommender System","date":"2021-06-16","arxiv_id":"2106.08700","repositories_listed":0,"syntology":null},{"url":null,"slug":"codert-distilling-encoder-representations","title":"CoDERT: Distilling Encoder Representations with Co-learning for Transducer-based Speech Recognition","date":"2021-06-14","arxiv_id":"2106.07734","repositories_listed":0,"syntology":null},{"url":null,"slug":"energy-efficient-knowledge-distillation-for","title":"Energy-efficient Knowledge Distillation for Spiking Neural Networks","date":"2021-06-14","arxiv_id":"2106.07172","repositories_listed":0,"syntology":null},{"url":"/paper/guiding-teacher-forcing-with-seer-forcing-for","slug":"guiding-teacher-forcing-with-seer-forcing-for","title":"Guiding Teacher Forcing with Seer Forcing for Neural Machine Translation","date":"2021-06-12","arxiv_id":"2106.06751","repositories_listed":0,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/guiding-teacher-forcing-with-seer-forcing-for#ran","syntology_url":"https://syntology.ai/paper/2106.06751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2106.06751"}},"official":null}},{"url":null,"slug":"refbert-compressing-bert-by-referencing-to","title":"RefBERT: Compressing BERT by Referencing to Pre-computed Representations","date":"2021-06-11","arxiv_id":"2106.08898","repositories_listed":0,"syntology":null},{"url":null,"slug":"graph-symbiosis-learning","title":"AKE-GNN: Effective Graph Learning with Adaptive Knowledge Exchange","date":"2021-06-10","arxiv_id":"2106.05455","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-by-distillation-a-self-supervised","title":"Learning by Distillation: A Self-Supervised Learning Framework for Optical Flow Estimation","date":"2021-06-08","arxiv_id":"2106.04195","repositories_listed":0,"syntology":null},{"url":null,"slug":"rosearch-search-for-robust-student","title":"RoSearch: Search for Robust Student Architectures When Distilling Pre-trained Language Models","date":"2021-06-07","arxiv_id":"2106.03613","repositories_listed":0,"syntology":null},{"url":null,"slug":"mergedistill-merging-pre-trained-language","title":"MergeDistill: Merging Pre-trained Language Models using Distillation","date":"2021-06-05","arxiv_id":"2106.02834","repositories_listed":0,"syntology":null},{"url":null,"slug":"not-all-knowledge-is-created-equal","title":"Not All Knowledge Is Created Equal: Mutual Distillation of Confident Knowledge","date":"2021-06-02","arxiv_id":"2106.01489","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-teacher-is-enough-pre-trained-language","title":"One Teacher is Enough? Pre-trained Language Model Distillation from Multiple Teachers","date":"2021-06-02","arxiv_id":"2106.01023","repositories_listed":0,"syntology":null},{"url":null,"slug":"claim-matching-beyond-english-to-scale-global","title":"Claim Matching Beyond English to Scale Global Fact-Checking","date":"2021-06-01","arxiv_id":"2106.00853","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-learning-for-neural-machine","title":"Continual Learning for Neural Machine Translation","date":"2021-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"cost-effective-deployment-of-bert-models-in-1","title":"Cost-effective Deployment of BERT Models in Serverless Environment","date":"2021-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"modality-specific-distillation-1","title":"Modality-specific Distillation","date":"2021-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-grained-knowledge-distillation-for","title":"Multi-Grained Knowledge Distillation for Named Entity Recognition","date":"2021-06-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"natural-statistics-of-network-activations-and","title":"Natural Statistics of Network Activations and Implications for Knowledge Distillation","date":"2021-06-01","arxiv_id":"2106.00368","repositories_listed":0,"syntology":null},{"url":null,"slug":"reinforced-iterative-knowledge-distillation","title":"Reinforced Iterative Knowledge Distillation for Cross-Lingual Named Entity Recognition","date":"2021-06-01","arxiv_id":"2106.00241","repositories_listed":0,"syntology":null},{"url":null,"slug":"fretal-generalizing-deepfake-detection-using","title":"FReTAL: Generalizing Deepfake Detection using Knowledge Distillation and Representation Learning","date":"2021-05-28","arxiv_id":"2105.13617","repositories_listed":0,"syntology":null},{"url":null,"slug":"fair-feature-distillation-for-visual","title":"Fair Feature Distillation for Visual Recognition","date":"2021-05-27","arxiv_id":"2106.04411","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-does-distilled-data-complexity-impact-the","title":"How Does Distilled Data Complexity Impact the Quality and Confidence of Non-Autoregressive Machine Translation?","date":"2021-05-27","arxiv_id":"2105.12900","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-detnas-upgrade-your-detector-with-nas","title":"Joint-DetNAS: Upgrade Your Detector with NAS, Pruning and Dynamic Distillation","date":"2021-05-27","arxiv_id":"2105.12971","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-understanding-knowledge-distillation","title":"Towards Understanding Knowledge Distillation","date":"2021-05-27","arxiv_id":"2105.13093","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowsr-knowledge-sharing-among-homogeneous","title":"KnowSR: Knowledge Sharing among Homogeneous Agents in Multi-agent Reinforcement Learning","date":"2021-05-25","arxiv_id":"2105.11611","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-monocular-depth-estimation-with","title":"Real-time Monocular Depth Estimation with Sparse Supervision on Mobile","date":"2021-05-25","arxiv_id":"2105.12053","repositories_listed":0,"syntology":null},{"url":null,"slug":"airnet-neural-network-transmission-over-the","title":"AirNet: Neural Network Transmission over the Air","date":"2021-05-24","arxiv_id":"2105.11166","repositories_listed":0,"syntology":null},{"url":null,"slug":"experimenting-with-knowledge-distillation","title":"Experimenting with Knowledge Distillation techniques for performing Brain Tumor Segmentation","date":"2021-05-24","arxiv_id":"2105.11486","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-knowledge-distillation-for-object","title":"Revisiting Knowledge Distillation for Object Detection","date":"2021-05-22","arxiv_id":"2105.10633","repositories_listed":0,"syntology":null},{"url":null,"slug":"inplace-knowledge-distillation-with-teacher","title":"Inplace knowledge distillation with teacher assistant for improved training of flexible deep neural networks","date":"2021-05-18","arxiv_id":"2105.08369","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-dense-video-captioning-via","title":"Weakly Supervised Dense Video Captioning via Jointly Usage of Knowledge Distillation and Cross-modal Matching","date":"2021-05-18","arxiv_id":"2105.08252","repositories_listed":0,"syntology":null},{"url":null,"slug":"class-incremental-few-shot-object-detection","title":"Class-Incremental Few-Shot Object Detection","date":"2021-05-17","arxiv_id":"2105.07637","repositories_listed":0,"syntology":null},{"url":null,"slug":"stacked-acoustic-and-textual-encoding","title":"Stacked Acoustic-and-Textual Encoding: Integrating the Pre-trained Models into Speech Translation Encoders","date":"2021-05-12","arxiv_id":"2105.05752","repositories_listed":0,"syntology":null},{"url":null,"slug":"test-time-adaptation-toward-personalized","title":"Test-Time Adaptation Toward Personalized Speech Enhancement: Zero-Shot Learning with Knowledge Distillation","date":"2021-05-08","arxiv_id":"2105.03544","repositories_listed":0,"syntology":null},{"url":null,"slug":"regression-bugs-are-in-your-model-measuring","title":"Regression Bugs Are In Your Model! Measuring, Reducing and Analyzing Regressions In NLP Model Updates","date":"2021-05-07","arxiv_id":"2105.03048","repositories_listed":0,"syntology":null},{"url":null,"slug":"black-box-dissector-towards-erasing-based","title":"Black-Box Dissector: Towards Erasing-based Hard-Label Model Stealing Attack","date":"2021-05-03","arxiv_id":"2105.00623","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-peek-into-the-reasoning-of-neural-networks","title":"A Peek Into the Reasoning of Neural Networks: Interpreting with Structural Visual Concepts","date":"2021-05-01","arxiv_id":"2105.00290","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-for-swedish-ner-models","title":"Knowledge Distillation for Swedish NER models: A Search for Performance and Efficiency","date":"2021-05-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"distilling-eeg-representations-via-capsules","title":"Distilling EEG Representations via Capsules for Affective Computing","date":"2021-04-30","arxiv_id":"2105.00104","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-relation-preserving-knowledge-1","title":"Semantic Relation Preserving Knowledge Distillation for Image-to-Image Translation","date":"2021-04-30","arxiv_id":"2104.15082","repositories_listed":0,"syntology":null},{"url":null,"slug":"spirit-distillation-a-model-compression","title":"Spirit Distillation: A Model Compression Method with Multi-domain Knowledge Transfer","date":"2021-04-29","arxiv_id":"2104.14696","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-distillation-with-batch-knowledge","title":"Self-distillation with Batch Knowledge Ensembling Improves ImageNet Classification","date":"2021-04-27","arxiv_id":"2104.13298","repositories_listed":0,"syntology":null},{"url":null,"slug":"extract-then-distill-efficient-and-effective","title":"Extract then Distill: Efficient and Effective Task-Agnostic BERT Distillation","date":"2021-04-24","arxiv_id":"2104.11928","repositories_listed":0,"syntology":null},{"url":null,"slug":"relational-subsets-knowledge-distillation-for","title":"Relational Subsets Knowledge Distillation for Long-tailed Retinal Diseases Recognition","date":"2021-04-22","arxiv_id":"2104.11057","repositories_listed":0,"syntology":null},{"url":null,"slug":"brittle-features-may-help-anomaly-detection","title":"Brittle Features May Help Anomaly Detection","date":"2021-04-21","arxiv_id":"2104.10453","repositories_listed":0,"syntology":null},{"url":null,"slug":"orderly-dual-teacher-knowledge-distillation","title":"Orderly Dual-Teacher Knowledge Distillation for Lightweight Human Pose Estimation","date":"2021-04-21","arxiv_id":"2104.10414","repositories_listed":0,"syntology":null},{"url":null,"slug":"edupal-leaves-no-professor-behind-supporting","title":"EduPal leaves no professor behind: Supporting faculty via a peer-powered recommender system","date":"2021-04-20","arxiv_id":"2104.12558","repositories_listed":0,"syntology":null},{"url":null,"slug":"compact-cnn-structure-learning-by-knowledge","title":"Compact CNN Structure Learning by Knowledge Distillation","date":"2021-04-19","arxiv_id":"2104.09191","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-learning-for-fake-audio-detection","title":"Continual Learning for Fake Audio Detection","date":"2021-04-15","arxiv_id":"2104.07286","repositories_listed":0,"syntology":null},{"url":"/paper/integration-of-pre-trained-networks-with","slug":"integration-of-pre-trained-networks-with","title":"Integration of Pre-trained Networks with Continuous Token Interface for End-to-End Spoken Language Understanding","date":"2021-04-15","arxiv_id":"2104.07253","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-learning-from-unlabeled-data-via","title":"Unsupervised Continual Learning Via Pseudo Labels","date":"2021-04-14","arxiv_id":"2104.07164","repositories_listed":0,"syntology":null},{"url":null,"slug":"sentence-embeddings-by-ensemble-distillation","title":"Sentence Embeddings by Ensemble Distillation","date":"2021-04-14","arxiv_id":"2104.06719","repositories_listed":0,"syntology":null},{"url":null,"slug":"dealing-with-missing-modalities-in-the-visual","title":"Dealing with Missing Modalities in the Visual Question Answer-Difference Prediction Task through Knowledge Distillation","date":"2021-04-13","arxiv_id":"2104.05965","repositories_listed":0,"syntology":null},{"url":null,"slug":"rankdistil-knowledge-distillation-for-ranking","title":"RankDistil: Knowledge Distillation for Ranking","date":"2021-04-13","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"source-and-target-bidirectional-knowledge","title":"Source and Target Bidirectional Knowledge Distillation for End-to-end Speech Translation","date":"2021-04-13","arxiv_id":"2104.06457","repositories_listed":0,"syntology":null},{"url":null,"slug":"dual-discriminator-adversarial-distillation","title":"Dual Discriminator Adversarial Distillation for Data-free Model Compression","date":"2021-04-12","arxiv_id":"2104.05382","repositories_listed":0,"syntology":null},{"url":null,"slug":"data-free-knowledge-distillation-with-soft","title":"Data-Free Knowledge Distillation with Soft Targeted Transfer Set Synthesis","date":"2021-04-10","arxiv_id":"2104.04868","repositories_listed":0,"syntology":null},{"url":null,"slug":"compressing-visual-linguistic-model-via","title":"Compressing Visual-linguistic Model via Knowledge Distillation","date":"2021-04-05","arxiv_id":"2104.02096","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-and-model-free-federated","title":"Decentralized and Model-Free Federated Learning: Consensus-Based Distillation in Function Space","date":"2021-04-01","arxiv_id":"2104.00352","repositories_listed":0,"syntology":null},{"url":null,"slug":"dialect-identification-through-adversarial","title":"Dialect Identification through Adversarial Learning and Knowledge Distillation on Romanian BERT","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"is-label-smoothing-truly-incompatible-with-1","title":"Is Label Smoothing Truly Incompatible with Knowledge Distillation: An Empirical Study","date":"2021-04-01","arxiv_id":"2104.00676","repositories_listed":0,"syntology":null},{"url":null,"slug":"topic-modeling-for-maternal-health-using","title":"Topic Modeling for Maternal Health Using Reddit","date":"2021-04-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"fixing-the-teacher-student-knowledge","title":"Fixing the Teacher-Student Knowledge Discrepancy in Distillation","date":"2021-03-31","arxiv_id":"2103.16844","repositories_listed":0,"syntology":null},{"url":null,"slug":"industry-scale-semi-supervised-learning-for","title":"Industry Scale Semi-Supervised Learning for Natural Language Understanding","date":"2021-03-29","arxiv_id":"2103.15871","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowru-knowledge-reusing-via-knowledge","title":"KnowRU: Knowledge Reusing via Knowledge Distillation in Multi-agent Reinforcement Learning","date":"2021-03-27","arxiv_id":"2103.14891","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-practical-survey-on-faster-and-lighter","title":"A Practical Survey on Faster and Lighter Transformers","date":"2021-03-26","arxiv_id":"2103.14636","repositories_listed":0,"syntology":null},{"url":null,"slug":"hands-on-guidance-for-distilling-object","title":"Hands-on Guidance for Distilling Object Detectors","date":"2021-03-26","arxiv_id":"2103.14337","repositories_listed":0,"syntology":null},{"url":null,"slug":"weakly-supervised-domain-adaptation-of-deep","title":"Weakly-Supervised Domain Adaptation of Deep Regression Trackers via Reinforced Knowledge Distillation","date":"2021-03-26","arxiv_id":"2103.14496","repositories_listed":0,"syntology":null},{"url":null,"slug":"spirit-distillation-precise-real-time","title":"Spirit Distillation: Precise Real-time Semantic Segmentation of Road Scenes with Insufficient Data","date":"2021-03-25","arxiv_id":"2103.13733","repositories_listed":0,"syntology":null},{"url":null,"slug":"balanced-softmax-cross-entropy-for","title":"Balanced softmax cross-entropy for incremental learning with and without memory","date":"2021-03-23","arxiv_id":"2103.12532","repositories_listed":0,"syntology":null},{"url":null,"slug":"student-network-learning-via-evolutionary","title":"Student Network Learning via Evolutionary Knowledge Distillation","date":"2021-03-23","arxiv_id":"2103.13811","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-nlp-cookbook-modern-recipes-for","title":"The NLP Cookbook: Modern Recipes for Transformer based Deep Learning Architectures","date":"2021-03-23","arxiv_id":"2104.10640","repositories_listed":0,"syntology":null},{"url":null,"slug":"compacting-deep-neural-networks-for-internet","title":"Compacting Deep Neural Networks for Internet of Things: Methods and Applications","date":"2021-03-20","arxiv_id":"2103.11083","repositories_listed":0,"syntology":null},{"url":null,"slug":"cost-effective-deployment-of-bert-models-in","title":"Cost-effective Deployment of BERT Models in Serverless Environment","date":"2021-03-19","arxiv_id":"2103.10673","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-knowledge-distillation-for","title":"Variational Knowledge Distillation for Disease Classification in Chest X-Rays","date":"2021-03-19","arxiv_id":"2103.10825","repositories_listed":0,"syntology":null},{"url":null,"slug":"similarity-transfer-for-knowledge","title":"Similarity Transfer for Knowledge Distillation","date":"2021-03-18","arxiv_id":"2103.10047","repositories_listed":0,"syntology":null}],"record_sha256":"35e38664eef72e7e433bbd57218c188ec88c5957e760e3b10c25f9780c945663","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}