{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/knowledge-distillation/papers/10","list_of":"/method/knowledge-distillation","method":"Knowledge Distillation","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":10,"pages_in_order":31,"rows_per_page":100,"rows":[901,1000],"of":3071,"counts":{"archive_papers_tagged":3071,"with_a_code_link":1258,"where_syntology_ran_a_sample":320,"not_listed_spam_title":0,"listed":3071,"listed_where_code_ran":320,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":276,"every_run_a_failure_of_syntologys_instrument":44,"listed_with_a_run_with_no_instrument_failure":276,"listed_every_run_a_failure_of_syntologys_instrument":44,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/knowledge-distillation","prev":"/method/knowledge-distillation/papers/9","next":"/method/knowledge-distillation/papers/11","papers":[{"paper":"/paper/an-efficient-end-to-end-approach-to-noise","slug":"an-efficient-end-to-end-approach-to-noise","title":"An Efficient End-to-End Approach to Noise Invariant Speech Features via Multi-Task Learning","date":"2024-03-13","arxiv_id":"2403.08654","n_code_links":1,"syntology":null},{"paper":null,"slug":"coronetgan-controlled-pruning-of-gans-via","title":"CoroNetGAN: Controlled Pruning of GANs via Hypernetworks","date":"2024-03-13","arxiv_id":"2403.08261","n_code_links":0,"syntology":null},{"paper":null,"slug":"distilling-named-entity-recognition-models","title":"Distilling Named Entity Recognition Models for Endangered Species from Large Language Models","date":"2024-03-13","arxiv_id":"2403.15430","n_code_links":0,"syntology":null},{"paper":null,"slug":"lix-implicitly-infusing-spatial-geometric","title":"LIX: Implicitly Infusing Spatial Geometric Prior Knowledge into Visual Semantic Segmentation for Autonomous Driving","date":"2024-03-13","arxiv_id":"2403.08215","n_code_links":0,"syntology":null},{"paper":null,"slug":"training-self-localization-models-for-unseen","title":"Training Self-localization Models for Unseen Unfamiliar Places via Teacher-to-Student Data-Free Knowledge Transfer","date":"2024-03-13","arxiv_id":"2403.10552","n_code_links":0,"syntology":null},{"paper":"/paper/continual-all-in-one-adverse-weather-removal","slug":"continual-all-in-one-adverse-weather-removal","title":"Continual All-in-One Adverse Weather Removal with Knowledge Replay on a Unified Network Structure","date":"2024-03-12","arxiv_id":"2403.07292","n_code_links":1,"syntology":null},{"paper":null,"slug":"distilling-the-knowledge-in-data-pruning","title":"Distilling the Knowledge in Data Pruning","date":"2024-03-12","arxiv_id":"2403.07854","n_code_links":0,"syntology":null},{"paper":"/paper/ediffiqa-towards-efficient-face-image-quality","slug":"ediffiqa-towards-efficient-face-image-quality","title":"eDifFIQA: Towards Efficient Face Image Quality Assessment Based On Denoising Diffusion Probabilistic Models","date":"2024-03-12","arxiv_id":null,"n_code_links":1,"syntology":null},{"paper":"/paper/microt-low-energy-and-adaptive-models-for","slug":"microt-low-energy-and-adaptive-models-for","title":"Low-Energy On-Device Personalization for MCUs","date":"2024-03-12","arxiv_id":"2403.08040","n_code_links":1,"syntology":null},{"paper":"/paper/answering-diverse-questions-via-text-attached","slug":"answering-diverse-questions-via-text-attached","title":"Answering Diverse Questions via Text Attached with Key Audio-Visual Clues","date":"2024-03-11","arxiv_id":"2403.06679","n_code_links":1,"syntology":null},{"paper":"/paper/aug-kd-anchor-based-mixup-generation-for-out","slug":"aug-kd-anchor-based-mixup-generation-for-out","title":"AuG-KD: Anchor-Based Mixup Generation for Out-of-Domain Knowledge Distillation","date":"2024-03-11","arxiv_id":"2403.07030","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":5,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 2 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ishikura-a/aug-kd"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":2,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"enhancing-adversarial-training-with-prior","title":"Enhancing Adversarial Training with Prior Knowledge Distillation for Robust Image Compression","date":"2024-03-11","arxiv_id":"2403.06700","n_code_links":0,"syntology":null},{"paper":null,"slug":"evolving-knowledge-distillation-with-large","title":"Evolving Knowledge Distillation with Large Language Models and Active Learning","date":"2024-03-11","arxiv_id":"2403.06414","n_code_links":0,"syntology":null},{"paper":"/paper/mend-meta-demonstration-distillation-for","slug":"mend-meta-demonstration-distillation-for","title":"MEND: Meta dEmonstratioN Distillation for Efficient and Effective In-Context Learning","date":"2024-03-11","arxiv_id":"2403.06914","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bigheiniu/mend"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"one-category-one-prompt-dataset-distillation","title":"One Category One Prompt: Dataset Distillation using Diffusion Models","date":"2024-03-11","arxiv_id":"2403.07142","n_code_links":0,"syntology":null},{"paper":"/paper/bit-mask-robust-contrastive-knowledge","slug":"bit-mask-robust-contrastive-knowledge","title":"Bit-mask Robust Contrastive Knowledge Distillation for Unsupervised Semantic Hashing","date":"2024-03-10","arxiv_id":"2403.06071","n_code_links":1,"syntology":null},{"paper":"/paper/cooperative-classification-and","slug":"cooperative-classification-and","title":"Cooperative Classification and Rationalization for Graph Generalization","date":"2024-03-10","arxiv_id":"2403.06239","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["yuelinan/codes-of-c2r"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/frequency-attention-for-knowledge","slug":"frequency-attention-for-knowledge","title":"Frequency Attention for Knowledge Distillation","date":"2024-03-09","arxiv_id":"2403.05894","n_code_links":1,"syntology":null},{"paper":"/paper/weakly-supervised-change-detection-via","slug":"weakly-supervised-change-detection-via","title":"Weakly Supervised Change Detection via Knowledge Distillation and Multiscale Sigmoid Inference","date":"2024-03-09","arxiv_id":"2403.05796","n_code_links":1,"syntology":null},{"paper":"/paper/radardistill-boosting-radar-based-object","slug":"radardistill-boosting-radar-based-object","title":"RadarDistill: Boosting Radar-based Object Detection Performance via Knowledge Distillation from LiDAR Features","date":"2024-03-08","arxiv_id":"2403.05061","n_code_links":1,"syntology":null},{"paper":null,"slug":"scene-graph-aided-radiology-report-generation","title":"Scene Graph Aided Radiology Report Generation","date":"2024-03-08","arxiv_id":"2403.05687","n_code_links":0,"syntology":null},{"paper":"/paper/a-study-of-dropout-induced-modality-bias-on","slug":"a-study-of-dropout-induced-modality-bias-on","title":"A Study of Dropout-Induced Modality Bias on Robustness to Missing Video Frames for Audio-Visual Speech Recognition","date":"2024-03-07","arxiv_id":"2403.04245","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-small-language-models-be-good-reasoners","title":"Can Small Language Models be Good Reasoners for Sequential Recommendation?","date":"2024-03-07","arxiv_id":"2403.04260","n_code_links":0,"syntology":null},{"paper":null,"slug":"mkf-ads-a-multi-knowledge-fused-anomaly","title":"MKF-ADS: Multi-Knowledge Fusion Based Self-supervised Anomaly Detection System for Control Area Network","date":"2024-03-07","arxiv_id":"2403.04293","n_code_links":0,"syntology":null},{"paper":null,"slug":"privacy-preserving-fine-tuning-of-large","title":"Privacy-preserving Fine-tuning of Large Language Models through Flatness","date":"2024-03-07","arxiv_id":"2403.04124","n_code_links":0,"syntology":null},{"paper":"/paper/self-adapting-large-visual-language-models-to","slug":"self-adapting-large-visual-language-models-to","title":"Self-Adapting Large Visual-Language Models to Edge Devices across Visual Modalities","date":"2024-03-07","arxiv_id":"2403.04908","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ramdrop/edgevl"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-teacher-free-graph-knowledge-distillation","slug":"a-teacher-free-graph-knowledge-distillation","title":"A Teacher-Free Graph Knowledge Distillation Framework with Dual Self-Distillation","date":"2024-03-06","arxiv_id":"2403.03483","n_code_links":1,"syntology":null},{"paper":"/paper/promptkd-unsupervised-prompt-distillation-for","slug":"promptkd-unsupervised-prompt-distillation-for","title":"PromptKD: Unsupervised Prompt Distillation for Vision-Language Models","date":"2024-03-05","arxiv_id":"2403.02781","n_code_links":1,"syntology":null},{"paper":null,"slug":"distilled-chatgpt-topic-sentiment-modeling","title":"Distilled ChatGPT Topic & Sentiment Modeling with Applications in Finance","date":"2024-03-04","arxiv_id":"2403.02185","n_code_links":0,"syntology":null},{"paper":null,"slug":"jep-kd-joint-embedding-predictive","title":"JEP-KD: Joint-Embedding Predictive Architecture Based Knowledge Distillation for Visual Speech Recognition","date":"2024-03-04","arxiv_id":"2403.18843","n_code_links":0,"syntology":null},{"paper":"/paper/powerskel-a-device-free-framework-using-csi","slug":"powerskel-a-device-free-framework-using-csi","title":"PowerSkel: A Device-Free Framework Using CSI Signal for Human Skeleton Estimation in Power Station","date":"2024-03-04","arxiv_id":"2403.01913","n_code_links":1,"syntology":null},{"paper":"/paper/align-to-distill-trainable-attention","slug":"align-to-distill-trainable-attention","title":"Align-to-Distill: Trainable Attention Alignment for Knowledge Distillation in Neural Machine Translation","date":"2024-03-03","arxiv_id":"2403.01479","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ncsoft/Align-to-Distill"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"hyperspectral-image-analysis-in-single-modal","title":"Hyperspectral Image Analysis in Single-Modal and Multimodal setting using Deep Learning Techniques","date":"2024-03-03","arxiv_id":"2403.01546","n_code_links":0,"syntology":null},{"paper":"/paper/logit-standardization-in-knowledge","slug":"logit-standardization-in-knowledge","title":"Logit Standardization in Knowledge Distillation","date":"2024-03-03","arxiv_id":"2403.01427","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sunshangquan/logit-standardardization-kd"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":null,"slug":"distilling-text-style-transfer-with-self","title":"Distilling Text Style Transfer With Self-Explanation From LLMs","date":"2024-03-02","arxiv_id":"2403.01106","n_code_links":0,"syntology":null},{"paper":"/paper/on-the-road-to-portability-compressing-end-to","slug":"on-the-road-to-portability-compressing-end-to","title":"On the Road to Portability: Compressing End-to-End Motion Planner for Autonomous Driving","date":"2024-03-02","arxiv_id":"2403.01238","n_code_links":1,"syntology":{"ran":8,"of":14,"n_ran_checked":6,"n_instrument":2,"unverified":6,"pointer_only":0,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","official":{"repos":["tulerfeng/PlanKD"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":5,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"data-efficient-event-camera-pre-training-via","title":"Data-efficient Event Camera Pre-training via Disentangled Masked Modeling","date":"2024-03-01","arxiv_id":"2403.00416","n_code_links":0,"syntology":null},{"paper":"/paper/differentially-private-knowledge-distillation","slug":"differentially-private-knowledge-distillation","title":"Differentially Private Knowledge Distillation via Synthetic Text Generation","date":"2024-03-01","arxiv_id":"2403.00932","n_code_links":1,"syntology":null},{"paper":"/paper/a-cognitive-based-trajectory-prediction","slug":"a-cognitive-based-trajectory-prediction","title":"A Cognitive-Based Trajectory Prediction Approach for Autonomous Driving","date":"2024-02-29","arxiv_id":"2402.19251","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-lightweight-low-light-image-enhancement","title":"A Lightweight Low-Light Image Enhancement Network via Channel Prior and Gamma Correction","date":"2024-02-28","arxiv_id":"2402.18147","n_code_links":0,"syntology":null},{"paper":null,"slug":"gradient-reweighting-towards-imbalanced-class","title":"Gradient Reweighting: Towards Imbalanced Class-Incremental Learning","date":"2024-02-28","arxiv_id":"2402.18528","n_code_links":0,"syntology":null},{"paper":"/paper/m3-vrd-multimodal-multi-task-multi-teacher","slug":"m3-vrd-multimodal-multi-task-multi-teacher","title":"3MVRD: Multimodal Multi-task Multi-teacher Visually-Rich Form Document Understanding","date":"2024-02-28","arxiv_id":"2402.17983","n_code_links":1,"syntology":null},{"paper":"/paper/sunshine-to-rainstorm-cross-weather-knowledge","slug":"sunshine-to-rainstorm-cross-weather-knowledge","title":"Sunshine to Rainstorm: Cross-Weather Knowledge Distillation for Robust 3D Object Detection","date":"2024-02-28","arxiv_id":"2402.18493","n_code_links":1,"syntology":null},{"paper":null,"slug":"mcf-vc-mitigate-catastrophic-forgetting-in","title":"MCF-VC: Mitigate Catastrophic Forgetting in Class-Incremental Learning for Multimodal Video Captioning","date":"2024-02-27","arxiv_id":"2402.17680","n_code_links":0,"syntology":null},{"paper":"/paper/promptmm-multi-modal-knowledge-distillation","slug":"promptmm-multi-modal-knowledge-distillation","title":"PromptMM: Multi-Modal Knowledge Distillation for Recommendation with Prompt-Tuning","date":"2024-02-27","arxiv_id":"2402.17188","n_code_links":1,"syntology":{"ran":10,"of":11,"n_ran_checked":9,"n_instrument":1,"unverified":1,"pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hkuds/promptmm"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"sddgr-stable-diffusion-based-deep-generative","title":"SDDGR: Stable Diffusion-based Deep Generative Replay for Class Incremental Object Detection","date":"2024-02-27","arxiv_id":"2402.17323","n_code_links":0,"syntology":null},{"paper":"/paper/sinkhorn-distance-minimization-for-knowledge","slug":"sinkhorn-distance-minimization-for-knowledge","title":"Sinkhorn Distance Minimization for Knowledge Distillation","date":"2024-02-27","arxiv_id":"2402.17110","n_code_links":1,"syntology":null},{"paper":null,"slug":"structural-teacher-student-normality-learning","title":"Structural Teacher-Student Normality Learning for Multi-Class Anomaly Detection and Localization","date":"2024-02-27","arxiv_id":"2402.17091","n_code_links":0,"syntology":null},{"paper":null,"slug":"dtcm-deep-transformer-capsule-mutual","title":"DTCM: Deep Transformer Capsule Mutual Distillation for Multivariate Time Series Classification","date":"2024-02-26","arxiv_id":null,"n_code_links":0,"syntology":null},{"paper":null,"slug":"llm-based-privacy-data-augmentation-guided-by","title":"LLM-based Privacy Data Augmentation Guided by Knowledge Distillation with a Distribution Tutor for Medical Text Classification","date":"2024-02-26","arxiv_id":"2402.16515","n_code_links":0,"syntology":null},{"paper":"/paper/llm-inference-unveiled-survey-and-roofline","slug":"llm-inference-unveiled-survey-and-roofline","title":"LLM Inference Unveiled: Survey and Roofline Model Insights","date":"2024-02-26","arxiv_id":"2402.16363","n_code_links":2,"syntology":{"ran":10,"of":10,"n_ran_checked":10,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hahnyuan/llm-viewer"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/m2mkd-module-to-module-knowledge-distillation","slug":"m2mkd-module-to-module-knowledge-distillation","title":"m2mKD: Module-to-Module Knowledge Distillation for Modular Transformers","date":"2024-02-26","arxiv_id":"2402.16918","n_code_links":1,"syntology":null},{"paper":null,"slug":"skill-similarity-aware-knowledge-distillation","title":"SKILL: Similarity-aware Knowledge distILLation for Speech Self-Supervised Learning","date":"2024-02-26","arxiv_id":"2402.16830","n_code_links":0,"syntology":null},{"paper":null,"slug":"enhancing-systematic-decompositional-natural","title":"Enhancing Systematic Decompositional Natural Language Inference Using Informal Logic","date":"2024-02-22","arxiv_id":"2402.14798","n_code_links":0,"syntology":null},{"paper":null,"slug":"practical-insights-into-knowledge","title":"Practical Insights into Knowledge Distillation for Pre-Trained Models","date":"2024-02-22","arxiv_id":"2402.14922","n_code_links":0,"syntology":null},{"paper":"/paper/tie-kd-teacher-independent-and-explainable","slug":"tie-kd-teacher-independent-and-explainable","title":"TIE-KD: Teacher-Independent and Explainable Knowledge Distillation for Monocular Depth Estimation","date":"2024-02-22","arxiv_id":"2402.14340","n_code_links":1,"syntology":null},{"paper":"/paper/packd-pattern-clustered-knowledge","slug":"packd-pattern-clustered-knowledge","title":"PaCKD: Pattern-Clustered Knowledge Distillation for Compressing Memory Access Prediction Models","date":"2024-02-21","arxiv_id":"2402.13441","n_code_links":1,"syntology":null},{"paper":null,"slug":"push-quantization-aware-training-toward-full","title":"In-Distribution Consistency Regularization Improves the Generalization of Quantization-Aware Training","date":"2024-02-21","arxiv_id":"2402.13497","n_code_links":0,"syntology":null},{"paper":null,"slug":"unsupervised-text-style-transfer-via-llms-and","title":"Unsupervised Text Style Transfer via LLMs and Attention Masking with Multi-way Interactions","date":"2024-02-21","arxiv_id":"2402.13647","n_code_links":0,"syntology":null},{"paper":null,"slug":"wisdom-of-committee-distilling-from","title":"Wisdom of Committee: Distilling from Foundation Model to Specialized Application Model","date":"2024-02-21","arxiv_id":"2402.14035","n_code_links":0,"syntology":null},{"paper":"/paper/a-survey-on-knowledge-distillation-of-large","slug":"a-survey-on-knowledge-distillation-of-large","title":"A Survey on Knowledge Distillation of Large Language Models","date":"2024-02-20","arxiv_id":"2402.13116","n_code_links":1,"syntology":null},{"paper":"/paper/improve-cross-architecture-generalization-on","slug":"improve-cross-architecture-generalization-on","title":"Improve Cross-Architecture Generalization on Dataset Distillation","date":"2024-02-20","arxiv_id":"2402.13007","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["distill-generalization-group/distill-generalization"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/promptkd-distilling-student-friendly","slug":"promptkd-distilling-student-friendly","title":"PromptKD: Distilling Student-Friendly Knowledge for Generative Language Models via Prompt Tuning","date":"2024-02-20","arxiv_id":"2402.12842","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":8,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["gmkim-ai/promptkd"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-the-byzantine-resilience-of-distillation","slug":"on-the-byzantine-resilience-of-distillation","title":"On the Byzantine-Resilience of Distillation-Based Federated Learning","date":"2024-02-19","arxiv_id":"2402.12265","n_code_links":1,"syntology":null},{"paper":"/paper/towards-cross-tokenizer-distillation-the","slug":"towards-cross-tokenizer-distillation-the","title":"Towards Cross-Tokenizer Distillation: the Universal Logit Distillation Loss for LLMs","date":"2024-02-19","arxiv_id":"2402.12030","n_code_links":3,"syntology":{"ran":14,"of":19,"n_ran_checked":14,"n_instrument":0,"unverified":5,"pointer_only":8,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["nicolas-bzrd/llm-recipes"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/teacher-as-a-lenient-expert-teacher-agnostic","slug":"teacher-as-a-lenient-expert-teacher-agnostic","title":"Teacher as a Lenient Expert: Teacher-Agnostic Data-Free Knowledge Distillation","date":"2024-02-18","arxiv_id":"2402.12406","n_code_links":1,"syntology":null},{"paper":"/paper/graphkd-exploring-knowledge-distillation","slug":"graphkd-exploring-knowledge-distillation","title":"GraphKD: Exploring Knowledge Distillation Towards Document Object Detection with Structured Graph Creation","date":"2024-02-17","arxiv_id":"2402.11401","n_code_links":1,"syntology":null},{"paper":"/paper/knowledge-distillation-based-on-transformed","slug":"knowledge-distillation-based-on-transformed","title":"Knowledge Distillation Based on Transformed Teacher Matching","date":"2024-02-17","arxiv_id":"2402.11148","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":2,"n_instrument":3,"unverified":2,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["zkxufo/TTM"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"on-good-practices-for-task-specific","title":"On Good Practices for Task-Specific Distillation of Large Pretrained Visual Models","date":"2024-02-17","arxiv_id":"2402.11305","n_code_links":0,"syntology":null},{"paper":"/paper/bitdistiller-unleashing-the-potential-of-sub","slug":"bitdistiller-unleashing-the-potential-of-sub","title":"BitDistiller: Unleashing the Potential of Sub-4-Bit LLMs via Self-Distillation","date":"2024-02-16","arxiv_id":"2402.10631","n_code_links":2,"syntology":{"ran":8,"of":15,"n_ran_checked":7,"n_instrument":1,"unverified":7,"pointer_only":3,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","official":{"repos":["dd-duda/bitdistiller"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"fedd2s-personalized-data-free-federated","title":"FedD2S: Personalized Data-Free Federated Knowledge Distillation","date":"2024-02-16","arxiv_id":"2402.10846","n_code_links":0,"syntology":null},{"paper":"/paper/incremental-sequence-labeling-a-tale-of-two","slug":"incremental-sequence-labeling-a-tale-of-two","title":"Incremental Sequence Labeling: A Tale of Two Shifts","date":"2024-02-16","arxiv_id":"2402.10447","n_code_links":2,"syntology":null},{"paper":"/paper/nuteprune-efficient-progressive-pruning-with","slug":"nuteprune-efficient-progressive-pruning-with","title":"NutePrune: Efficient Progressive Pruning with Numerous Teachers for Large Language Models","date":"2024-02-15","arxiv_id":"2402.09773","n_code_links":1,"syntology":{"ran":9,"of":16,"n_ran_checked":5,"n_instrument":4,"unverified":7,"pointer_only":16,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 4 where Syntology's instrument failed) · 7 unverified","official":{"repos":["lucius-lsr/nuteprune"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"walsh-domain-neural-network-for-power","title":"Walsh-domain Neural Network for Power Amplifier Behavioral Modelling and Digital Predistortion","date":"2024-02-15","arxiv_id":"2402.09964","n_code_links":0,"syntology":null},{"paper":"/paper/fedsikd-clients-similarity-and-knowledge","slug":"fedsikd-clients-similarity-and-knowledge","title":"FedSiKD: Clients Similarity and Knowledge Distillation: Addressing Non-i.i.d. and Constraints in Federated Learning","date":"2024-02-14","arxiv_id":"2402.09095","n_code_links":1,"syntology":null},{"paper":null,"slug":"integrating-chatgpt-into-secure-hospital","title":"Integrating ChatGPT into Secure Hospital Networks: A Case Study on Improving Radiology Report Analysis","date":"2024-02-14","arxiv_id":"2402.09358","n_code_links":0,"syntology":null},{"paper":"/paper/training-heterogeneous-client-models-using","slug":"training-heterogeneous-client-models-using","title":"Training Heterogeneous Client Models using Knowledge Distillation in Serverless Federated Learning","date":"2024-02-11","arxiv_id":"2402.07295","n_code_links":1,"syntology":null},{"paper":null,"slug":"two-stage-multi-task-self-supervised-learning","title":"Two-Stage Multi-task Self-Supervised Learning for Medical Image Segmentation","date":"2024-02-11","arxiv_id":"2402.07119","n_code_links":0,"syntology":null},{"paper":"/paper/domain-adaptable-fine-tune-distillation","slug":"domain-adaptable-fine-tune-distillation","title":"Domain Adaptable Fine-Tune Distillation Framework For Advancing Farm Surveillance","date":"2024-02-10","arxiv_id":"2402.07059","n_code_links":1,"syntology":null},{"paper":null,"slug":"embedding-compression-for-teacher-to-student","title":"Embedding Compression for Teacher-to-Student Knowledge Transfer","date":"2024-02-09","arxiv_id":"2402.06761","n_code_links":0,"syntology":null},{"paper":"/paper/multi-source-free-domain-adaptation-via","slug":"multi-source-free-domain-adaptation-via","title":"Multi-source-free Domain Adaptation via Uncertainty-aware Adaptive Distillation","date":"2024-02-09","arxiv_id":"2402.06213","n_code_links":1,"syntology":null},{"paper":null,"slug":"large-language-model-meets-graph-neural","title":"Large Language Model Meets Graph Neural Network in Knowledge Distillation","date":"2024-02-08","arxiv_id":"2402.05894","n_code_links":0,"syntology":null},{"paper":"/paper/efficientvit-sam-accelerated-segment-anything","slug":"efficientvit-sam-accelerated-segment-anything","title":"EfficientViT-SAM: Accelerated Segment Anything Model Without Accuracy Loss","date":"2024-02-07","arxiv_id":"2402.05008","n_code_links":1,"syntology":null},{"paper":null,"slug":"knowledge-distillation-for-road-detection","title":"Knowledge Distillation for Road Detection based on cross-model Semi-Supervised Learning","date":"2024-02-07","arxiv_id":"2402.05305","n_code_links":0,"syntology":null},{"paper":"/paper/tinyllm-learning-a-small-student-from","slug":"tinyllm-learning-a-small-student-from","title":"Beyond Answers: Transferring Reasoning Capabilities to Smaller LLMs Using Multi-Teacher Knowledge Distillation","date":"2024-02-07","arxiv_id":"2402.04616","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yikunhan42/tinyllm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/vision-superalignment-weak-to-strong","slug":"vision-superalignment-weak-to-strong","title":"Vision Superalignment: Weak-to-Strong Generalization for Vision Foundation Models","date":"2024-02-06","arxiv_id":"2402.03749","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":2,"n_instrument":2,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ggjy/vision_weak_to_strong"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/dual-knowledge-distillation-for-efficient","slug":"dual-knowledge-distillation-for-efficient","title":"Dual Knowledge Distillation for Efficient Sound Event Detection","date":"2024-02-05","arxiv_id":"2402.02781","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-model-distilling-medication","slug":"large-language-model-distilling-medication","title":"Large Language Model Distilling Medication Recommendation Model","date":"2024-02-05","arxiv_id":"2402.02803","n_code_links":1,"syntology":null},{"paper":null,"slug":"bi-cryptonets-leveraging-different-level","title":"Bi-CryptoNets: Leveraging Different-Level Privacy for Encrypted Inference","date":"2024-02-02","arxiv_id":"2402.01296","n_code_links":0,"syntology":null},{"paper":"/paper/cascaded-scaling-classifier-class-incremental","slug":"cascaded-scaling-classifier-class-incremental","title":"Class incremental learning with probability dampening and cascaded gated classifier","date":"2024-02-02","arxiv_id":"2402.01262","n_code_links":2,"syntology":null},{"paper":"/paper/cooperative-knowledge-distillation-a-learner","slug":"cooperative-knowledge-distillation-a-learner","title":"Cooperative Knowledge Distillation: A Learner Agnostic Approach","date":"2024-02-02","arxiv_id":"2402.05942","n_code_links":1,"syntology":null},{"paper":null,"slug":"faster-inference-of-integer-swin-transformer","title":"Faster Inference of Integer SWIN Transformer by Removing the GELU Activation","date":"2024-02-02","arxiv_id":"2402.01169","n_code_links":0,"syntology":null},{"paper":null,"slug":"spiking-centernet-a-distillation-boosted","title":"Spiking CenterNet: A Distillation-boosted Spiking Neural Network for Object Detection","date":"2024-02-02","arxiv_id":"2402.01287","n_code_links":0,"syntology":null},{"paper":null,"slug":"addressing-bias-through-ensemble-learning-and","title":"Addressing Bias Through Ensemble Learning and Regularized Fine-Tuning","date":"2024-02-01","arxiv_id":"2402.00910","n_code_links":0,"syntology":null},{"paper":null,"slug":"distilling-conditional-diffusion-models-for","title":"Augmenting Offline Reinforcement Learning with State-only Interactions","date":"2024-02-01","arxiv_id":"2402.00807","n_code_links":0,"syntology":null},{"paper":null,"slug":"dual-student-knowledge-distillation-networks","title":"Dual-Student Knowledge Distillation Networks for Unsupervised Anomaly Detection","date":"2024-02-01","arxiv_id":"2402.00448","n_code_links":0,"syntology":null},{"paper":null,"slug":"epsd-early-pruning-with-self-distillation-for","title":"EPSD: Early Pruning with Self-Distillation for Efficient Model Compression","date":"2024-01-31","arxiv_id":"2402.00084","n_code_links":0,"syntology":null},{"paper":null,"slug":"scavenging-hyena-distilling-transformers-into","title":"Scavenging Hyena: Distilling Transformers into Long Convolution Models","date":"2024-01-31","arxiv_id":"2401.17574","n_code_links":0,"syntology":null},{"paper":"/paper/llamp-large-language-model-made-powerful-for","slug":"llamp-large-language-model-made-powerful-for","title":"LLaMP: Large Language Model Made Powerful for High-fidelity Materials Knowledge Retrieval and Distillation","date":"2024-01-30","arxiv_id":"2401.17244","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chiang-yuan/llamp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"distilling-privileged-multimodal-information","title":"Distilling Privileged Multimodal Information for Expression Recognition using Optimal Transport","date":"2024-01-27","arxiv_id":"2401.15489","n_code_links":0,"syntology":null}],"record_sha256":"c9b0d22100fb8cc8ae4e0a28f0e76911afacd2dae2b574d889fbdd22c861e7f2","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}