{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/model-compression/papers/6","list_of":"/task/model-compression","task":"Model Compression","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":6,"pages_in_order":14,"rows_per_page":100,"rows":[501,600],"of":1356,"counts":{"archive_papers_tagged":1356,"with_a_code_link":440,"where_syntology_ran_a_sample":119,"not_listed_spam_title":0,"listed":1356,"listed_where_code_ran":119,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":97,"every_run_a_failure_of_syntologys_instrument":22,"listed_with_a_run_with_no_instrument_failure":97,"listed_every_run_a_failure_of_syntologys_instrument":22,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/model-compression","prev":"/task/model-compression/papers/5","next":"/task/model-compression/papers/7","papers":[{"url":null,"slug":"iterabre-iterative-recovery-aided-block","title":"IteRABRe: Iterative Recovery-Aided Block Reduction","date":"2025-03-08","arxiv_id":"2503.06291","repositories_listed":0,"syntology":null},{"url":null,"slug":"lvlm-compress-bench-benchmarking-the-broader","title":"LVLM-Compress-Bench: Benchmarking the Broader Impact of Large Vision-Language Model Compression","date":"2025-03-06","arxiv_id":"2503.04982","repositories_listed":0,"syntology":null},{"url":null,"slug":"tinyr1-32b-preview-boosting-accuracy-with","title":"TinyR1-32B-Preview: Boosting Accuracy with Branch-Merge Distillation","date":"2025-03-06","arxiv_id":"2503.04872","repositories_listed":0,"syntology":null},{"url":null,"slug":"10k-is-enough-an-ultra-lightweight-binarized","title":"10K is Enough: An Ultra-Lightweight Binarized Network for Infrared Small-Target Detection","date":"2025-03-04","arxiv_id":"2503.02662","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-the-tip-of-efficiency-uncovering-the","title":"Beyond the Tip of Efficiency: Uncovering the Submerged Threats of Jailbreak Attacks in Small Language Models","date":"2025-02-27","arxiv_id":"2502.19883","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-transformers-on-the-edge-a","title":"Vision Transformers on the Edge: A Comprehensive Survey of Model Compression and Acceleration Strategies","date":"2025-02-26","arxiv_id":"2503.02891","repositories_listed":0,"syntology":null},{"url":null,"slug":"afroxlmr-comet-multilingual-knowledge","title":"AfroXLMR-Comet: Multilingual Knowledge Distillation with Attention Matching for Low-Resource languages","date":"2025-02-25","arxiv_id":"2502.18020","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-lottery-llm-hypothesis-rethinking-what","title":"The Lottery LLM Hypothesis, Rethinking What Abilities Should LLM Compression Preserve?","date":"2025-02-24","arxiv_id":"2502.17535","repositories_listed":0,"syntology":null},{"url":null,"slug":"swallowing-the-poison-pills-insights-from","title":"Swallowing the Poison Pills: Insights from Vulnerability Disparity Among LLMs","date":"2025-02-23","arxiv_id":"2502.18518","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-compression-meets-model-compression","title":"When Compression Meets Model Compression: Memory-Efficient Double Compression for Large Language Models","date":"2025-02-21","arxiv_id":"2502.15443","repositories_listed":0,"syntology":null},{"url":"/paper/efficient-ai-in-practice-training-and","slug":"efficient-ai-in-practice-training-and","title":"Efficient AI in Practice: Training and Deployment of Efficient LLMs for Industry Applications","date":"2025-02-20","arxiv_id":"2502.14305","repositories_listed":0,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-ai-in-practice-training-and#ran","syntology_url":"https://syntology.ai/paper/2502.14305","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14305"}},"official":null}},{"url":null,"slug":"optimizing-singular-spectrum-for-large","title":"Optimizing Singular Spectrum for Large Language Model Compression","date":"2025-02-20","arxiv_id":"2502.15092","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-foundation-models-in-medical-image","title":"Vision Foundation Models in Medical Image Analysis: Advances and Challenges","date":"2025-02-20","arxiv_id":"2502.14584","repositories_listed":0,"syntology":null},{"url":null,"slug":"maskprune-mask-based-llm-pruning-for-layer","title":"MaskPrune: Mask-based LLM Pruning for Layer-wise Uniform Structures","date":"2025-02-19","arxiv_id":"2502.14008","repositories_listed":0,"syntology":null},{"url":null,"slug":"every-expert-matters-towards-effective","title":"Every Expert Matters: Towards Effective Knowledge Distillation for Mixture-of-Experts Language Models","date":"2025-02-18","arxiv_id":"2502.12947","repositories_listed":0,"syntology":null},{"url":null,"slug":"optishear-towards-efficient-and-adaptive","title":"OPTISHEAR: Towards Efficient and Adaptive Pruning of Large Language Models via Evolutionary Optimization","date":"2025-02-15","arxiv_id":"2502.10735","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-language-models-for-edge-networks-a","title":"Vision-Language Models for Edge Networks: A Comprehensive Survey","date":"2025-02-11","arxiv_id":"2502.07855","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-rank-compression-for-imc-arrays","title":"Low-Rank Compression for IMC Arrays","date":"2025-02-10","arxiv_id":"2502.07820","repositories_listed":0,"syntology":null},{"url":null,"slug":"runtime-tunable-tsetlin-machines-for-edge","title":"Runtime Tunable Tsetlin Machines for Edge Inference on eFPGAs","date":"2025-02-10","arxiv_id":"2502.07823","repositories_listed":0,"syntology":null},{"url":null,"slug":"synergistic-effects-of-knowledge-distillation","title":"Synergistic Effects of Knowledge Distillation and Structured Pruning for Self-Supervised Speech Models","date":"2025-02-09","arxiv_id":"2502.05837","repositories_listed":0,"syntology":null},{"url":null,"slug":"theoretical-guarantees-for-low-rank","title":"Theoretical Guarantees for Low-Rank Compression of Deep Neural Networks","date":"2025-02-04","arxiv_id":"2502.02766","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerating-linear-recurrent-neural-networks","title":"Accelerating Linear Recurrent Neural Networks for the Edge with Unstructured Sparsity","date":"2025-02-03","arxiv_id":"2502.01330","repositories_listed":0,"syntology":null},{"url":null,"slug":"mind-modality-informed-knowledge-distillation","title":"MIND: Modality-Informed Knowledge Distillation Framework for Multimodal Clinical Prediction Tasks","date":"2025-02-03","arxiv_id":"2502.01158","repositories_listed":0,"syntology":null},{"url":null,"slug":"attention-sinks-and-outlier-features-a-catch","title":"Attention Sinks and Outlier Features: A 'Catch, Tag, and Release' Mechanism for Embeddings","date":"2025-02-02","arxiv_id":"2502.00919","repositories_listed":0,"syntology":null},{"url":null,"slug":"huff-llm-end-to-end-lossless-compression-for","title":"Huff-LLM: End-to-End Lossless Compression for Efficient LLM Inference","date":"2025-02-02","arxiv_id":"2502.00922","repositories_listed":0,"syntology":null},{"url":null,"slug":"role-of-mixup-in-topological-persistence","title":"Role of Mixup in Topological Persistence Based Knowledge Distillation for Wearable Sensor Data","date":"2025-02-02","arxiv_id":"2502.00779","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-supernet-training-with-orthogonal","title":"Efficient Supernet Training with Orthogonal Softmax for Scalable ASR Model Compression","date":"2025-01-31","arxiv_id":"2501.18895","repositories_listed":0,"syntology":null},{"url":null,"slug":"pivoting-factorization-a-compact-meta-low","title":"Pivoting Factorization: A Compact Meta Low-Rank Representation of Sparsity for Efficient Inference in Large Language Models","date":"2025-01-31","arxiv_id":"2501.19090","repositories_listed":0,"syntology":null},{"url":null,"slug":"taid-temporally-adaptive-interpolated","title":"TAID: Temporally Adaptive Interpolated Distillation for Efficient Knowledge Transfer in Language Models","date":"2025-01-28","arxiv_id":"2501.16937","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-accelerating-edge-ai-optimizing-resource","title":"On Accelerating Edge AI: Optimizing Resource-Constrained Environments","date":"2025-01-25","arxiv_id":"2501.15014","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-only-prune-once-designing-calibration","title":"You Only Prune Once: Designing Calibration-Free Model Compression With Policy Learning","date":"2025-01-25","arxiv_id":"2501.15296","repositories_listed":0,"syntology":null},{"url":null,"slug":"hwpq-hessian-free-weight-pruning-quantization","title":"SwiftPrune: Hessian-Free Weight Pruning for Large Language Models","date":"2025-01-24","arxiv_id":"2501.16376","repositories_listed":0,"syntology":null},{"url":null,"slug":"practical-quantum-federated-learning-and-its","title":"Practical quantum federated learning and its experimental demonstration","date":"2025-01-22","arxiv_id":"2501.12709","repositories_listed":0,"syntology":null},{"url":null,"slug":"atleus-accelerating-transformers-on-the-edge","title":"Atleus: Accelerating Transformers on the Edge Enabled by 3D Heterogeneous Manycore Architectures","date":"2025-01-16","arxiv_id":"2501.09588","repositories_listed":0,"syntology":null},{"url":null,"slug":"fasp-fast-and-accurate-structured-pruning-of","title":"FASP: Fast and Accurate Structured Pruning of Large Language Models","date":"2025-01-16","arxiv_id":"2501.09412","repositories_listed":0,"syntology":null},{"url":null,"slug":"knowledge-distillation-for-image-restoration","title":"Knowledge Distillation for Image Restoration : Simultaneous Learning from Degraded and Clean Images","date":"2025-01-16","arxiv_id":"2501.09268","repositories_listed":0,"syntology":null},{"url":null,"slug":"swsc-shared-weight-for-similar-channel-in-llm","title":"SWSC: Shared Weight for Similar Channel in LLM","date":"2025-01-15","arxiv_id":"2501.08631","repositories_listed":0,"syntology":null},{"url":null,"slug":"curing-large-models-compression-via-cur","title":"CURing Large Models: Compression via CUR Decomposition","date":"2025-01-08","arxiv_id":"2501.04211","repositories_listed":0,"syntology":null},{"url":null,"slug":"upaq-a-framework-for-real-time-and-energy","title":"UPAQ: A Framework for Real-Time and Energy-Efficient 3D Object Detection in Autonomous Vehicles","date":"2025-01-08","arxiv_id":"2501.04213","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-and-efficient-mixed-precision","title":"Effective and Efficient Mixed Precision Quantization of Speech Foundation Models","date":"2025-01-07","arxiv_id":"2501.03643","repositories_listed":0,"syntology":null},{"url":null,"slug":"strategic-fusion-optimizes-transformer","title":"Strategic Fusion Optimizes Transformer Compression","date":"2025-01-05","arxiv_id":"2501.03273","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-small-language-models-for-in","title":"Optimizing Small Language Models for In-Vehicle Function-Calling","date":"2025-01-04","arxiv_id":"2501.02342","repositories_listed":0,"syntology":null},{"url":null,"slug":"once-tuning-multiple-variants-tuning-once-and","title":"Once-Tuning-Multiple-Variants: Tuning Once and Expanded as Multiple Vision-Language Model Variants","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"random-conditioning-for-diffusion-model","title":"Random Conditioning for Diffusion Model Compression with Distillation","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-acoustic-scene-classification-in","title":"Improving Acoustic Scene Classification in Low-Resource Conditions","date":"2024-12-30","arxiv_id":"2412.20722","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-alignment-based-knowledge","title":"Feature Alignment-Based Knowledge Distillation for Efficient Compression of Large Language Models","date":"2024-12-27","arxiv_id":"2412.19449","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimization-and-scalability-of-collaborative","title":"Optimization and Scalability of Collaborative Filtering Algorithms in Large Language Models","date":"2024-12-25","arxiv_id":"2412.18715","repositories_listed":0,"syntology":null},{"url":null,"slug":"htr-jand-handwritten-text-recognition-with","title":"HTR-JAND: Handwritten Text Recognition with Joint Attention Network and Knowledge Distillation","date":"2024-12-24","arxiv_id":"2412.18524","repositories_listed":0,"syntology":null},{"url":null,"slug":"cosurfgs-collaborative-3d-surface-gaussian","title":"CoSurfGS:Collaborative 3D Surface Gaussian Splatting with Distributed Learning for Large Scene Reconstruction","date":"2024-12-23","arxiv_id":"2412.17612","repositories_listed":0,"syntology":null},{"url":null,"slug":"edge-ai-for-agriculture-lightweight-vision","title":"Edge-AI for Agriculture: Lightweight Vision Models for Disease Detection in Resource-Limited Settings","date":"2024-12-23","arxiv_id":"2412.18635","repositories_listed":0,"syntology":null},{"url":null,"slug":"gqsa-group-quantization-and-sparsity-for","title":"GQSA: Group Quantization and Sparsity for Accelerating Large Language Model Inference","date":"2024-12-23","arxiv_id":"2412.17560","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-design-and-optimization-methods","title":"Lightweight Design and Optimization methods for DCNNs: Progress and Futures","date":"2024-12-22","arxiv_id":"2412.16886","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantics-prompting-data-free-quantization","title":"Semantics Prompting Data-Free Quantization for Low-Bit Vision Transformers","date":"2024-12-21","arxiv_id":"2412.16553","repositories_listed":0,"syntology":null},{"url":null,"slug":"deploying-foundation-model-powered-agent","title":"Deploying Foundation Model Powered Agent Services: A Survey","date":"2024-12-18","arxiv_id":"2412.13437","repositories_listed":0,"syntology":null},{"url":null,"slug":"trimllm-progressive-layer-dropping-for-domain","title":"TrimLLM: Progressive Layer Dropping for Domain-Specific LLMs","date":"2024-12-15","arxiv_id":"2412.11242","repositories_listed":0,"syntology":null},{"url":null,"slug":"activation-sparsity-opportunities-for","title":"Activation Sparsity Opportunities for Compressing General Large Language Models","date":"2024-12-13","arxiv_id":"2412.12178","repositories_listed":0,"syntology":null},{"url":null,"slug":"can-students-beyond-the-teacher-distilling","title":"Can Students Beyond The Teacher? Distilling Knowledge from Teacher's Bias","date":"2024-12-13","arxiv_id":"2412.09874","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimising-tinyml-with-quantization-and","title":"Optimising TinyML with Quantization and Distillation of Transformer and Mamba Models for Indoor Localisation on Edge Devices","date":"2024-12-12","arxiv_id":"2412.09289","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-rank-correction-for-quantized-llms","title":"Low-Rank Correction for Quantized LLMs","date":"2024-12-10","arxiv_id":"2412.07902","repositories_listed":0,"syntology":null},{"url":null,"slug":"compression-for-better-a-general-and-stable","title":"Compression for Better: A General and Stable Lossless Compression Framework","date":"2024-12-09","arxiv_id":"2412.06868","repositories_listed":0,"syntology":null},{"url":null,"slug":"lossless-model-compression-via-joint-low-rank","title":"Lossless Model Compression via Joint Low-Rank Factorization Optimization","date":"2024-12-09","arxiv_id":"2412.06867","repositories_listed":0,"syntology":null},{"url":null,"slug":"vq4all-efficient-neural-network","title":"VQ4ALL: Efficient Neural Network Representation via a Universal Codebook","date":"2024-12-09","arxiv_id":"2412.06875","repositories_listed":0,"syntology":null},{"url":null,"slug":"trimming-down-large-spiking-vision","title":"Trimming Down Large Spiking Vision Transformers via Heterogeneous Quantization Search","date":"2024-12-07","arxiv_id":"2412.05505","repositories_listed":0,"syntology":null},{"url":null,"slug":"cptquant-a-novel-mixed-precision-post","title":"CPTQuant -- A Novel Mixed Precision Post-Training Quantization Techniques for Large Language Models","date":"2024-12-03","arxiv_id":"2412.03599","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-model-compression-techniques-with","title":"Efficient Model Compression Techniques with FishLeg","date":"2024-12-03","arxiv_id":"2412.02328","repositories_listed":0,"syntology":null},{"url":null,"slug":"individual-content-and-motion-dynamics","title":"Individual Content and Motion Dynamics Preserved Pruning for Video Diffusion Models","date":"2024-11-27","arxiv_id":"2411.18375","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-pruning-of-text-to-image-models","title":"Efficient Pruning of Text-to-Image Models: Insights from Pruning Stable Diffusion","date":"2024-11-22","arxiv_id":"2411.15113","repositories_listed":0,"syntology":null},{"url":"/paper/taq-dit-time-aware-quantization-for-diffusion","slug":"taq-dit-time-aware-quantization-for-diffusion","title":"TaQ-DiT: Time-aware Quantization for Diffusion Transformers","date":"2024-11-21","arxiv_id":"2411.14172","repositories_listed":0,"syntology":{"n":7,"n_ran":5,"n_constructed":2,"n_ran_checked":2,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/taq-dit-time-aware-quantization-for-diffusion#ran","syntology_url":"https://syntology.ai/paper/2411.14172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14172"}},"official":null}},{"url":null,"slug":"fastnav-fine-tuned-adaptive-small-language","title":"FASTNav: Fine-tuned Adaptive Small-language-models Trained for Multi-point Robot Navigation","date":"2024-11-20","arxiv_id":"2411.13262","repositories_listed":0,"syntology":null},{"url":null,"slug":"puppet-cnn-input-adaptive-convolutional","title":"Puppet-CNN: Input-Adaptive Convolutional Neural Networks with Model Compression using Ordinary Differential Equation","date":"2024-11-19","arxiv_id":"2411.12876","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-makes-a-good-dataset-for-knowledge","title":"What Makes a Good Dataset for Knowledge Distillation?","date":"2024-11-19","arxiv_id":"2411.12817","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-resource-gap-deploying-advanced","title":"Bridging the Resource Gap: Deploying Advanced Imitation Learning Models onto Affordable Embedded Platforms","date":"2024-11-18","arxiv_id":"2411.11406","repositories_listed":0,"syntology":null},{"url":null,"slug":"re-parameterization-of-lightweight","title":"Re-Parameterization of Lightweight Transformer for On-Device Speech Emotion Recognition","date":"2024-11-14","arxiv_id":"2411.09339","repositories_listed":0,"syntology":null},{"url":null,"slug":"aser-activation-smoothing-and-error","title":"ASER: Activation Smoothing and Error Reconstruction for Large Language Model Quantization","date":"2024-11-12","arxiv_id":"2411.07762","repositories_listed":0,"syntology":null},{"url":null,"slug":"feature-interaction-fusion-self-distillation","title":"Feature Interaction Fusion Self-Distillation Network For CTR Prediction","date":"2024-11-12","arxiv_id":"2411.07508","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-traffic-signal-control-using-high","title":"Optimizing Traffic Signal Control using High-Dimensional State Representation and Efficient Deep Reinforcement Learning","date":"2024-11-12","arxiv_id":"2411.07759","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-word-vectors-to-multimodal-embeddings","title":"From Word Vectors to Multimodal Embeddings: Techniques, Applications, and Future Directions For Large Language Models","date":"2024-11-06","arxiv_id":"2411.05036","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-model-compression-for-bayesian","title":"Efficient Model Compression for Bayesian Neural Networks","date":"2024-11-01","arxiv_id":"2411.00273","repositories_listed":0,"syntology":null},{"url":"/paper/eora-training-free-compensation-for","slug":"eora-training-free-compensation-for","title":"EoRA: Training-free Compensation for Compressed LLM with Eigenspace Low-Rank Approximation","date":"2024-10-28","arxiv_id":"2410.21271","repositories_listed":0,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/eora-training-free-compensation-for#ran","syntology_url":"https://syntology.ai/paper/2410.21271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21271"}},"official":null}},{"url":null,"slug":"a-survey-of-small-language-models","title":"A Survey of Small Language Models","date":"2024-10-25","arxiv_id":"2410.20011","repositories_listed":0,"syntology":null},{"url":null,"slug":"switch-studying-with-teacher-for-knowledge","title":"SWITCH: Studying with Teacher for Knowledge Distillation of Large Language Models","date":"2024-10-25","arxiv_id":"2410.19503","repositories_listed":0,"syntology":null},{"url":null,"slug":"beware-of-calibration-data-for-pruning-large","title":"Beware of Calibration Data for Pruning Large Language Models","date":"2024-10-23","arxiv_id":"2410.17711","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-effective-data-free-knowledge","title":"Towards Effective Data-Free Knowledge Distillation via Diverse Diffusion Augmentation","date":"2024-10-23","arxiv_id":"2410.17606","repositories_listed":0,"syntology":null},{"url":null,"slug":"self-calibration-for-language-model","title":"Self-calibration for Language Model Quantization and Pruning","date":"2024-10-22","arxiv_id":"2410.17170","repositories_listed":0,"syntology":null},{"url":null,"slug":"identifying-sub-networks-in-neural-networks","title":"Identifying Sub-networks in Neural Networks via Functionally Similar Representations","date":"2024-10-21","arxiv_id":"2410.16484","repositories_listed":0,"syntology":null},{"url":null,"slug":"preview-based-category-contrastive-learning","title":"Preview-based Category Contrastive Learning for Knowledge Distillation","date":"2024-10-18","arxiv_id":"2410.14143","repositories_listed":0,"syntology":null},{"url":null,"slug":"crossquant-a-post-training-quantization","title":"CrossQuant: A Post-Training Quantization Method with Smaller Quantization Kernel for Precise Large Language Model Compression","date":"2024-10-10","arxiv_id":"2410.07505","repositories_listed":0,"syntology":null},{"url":null,"slug":"what-is-left-after-distillation-how-knowledge","title":"What is Left After Distillation? How Knowledge Transfer Impacts Fairness and Bias","date":"2024-10-10","arxiv_id":"2410.08407","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-compression-with-neural-architecture","title":"Large Language Model Compression with Neural Architecture Search","date":"2024-10-09","arxiv_id":"2410.06479","repositories_listed":0,"syntology":null},{"url":null,"slug":"spallm-unified-compressive-adaptation-of","title":"SpaLLM: Unified Compressive Adaptation of Large Language Models with Sketching","date":"2024-10-08","arxiv_id":"2410.06364","repositories_listed":0,"syntology":null},{"url":null,"slug":"espace-dimensionality-reduction-of","title":"ESPACE: Dimensionality Reduction of Activations for Model Compression","date":"2024-10-07","arxiv_id":"2410.05437","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-approximations-for-improving","title":"Continuous Approximations for Improving Quantization Aware Training of LLMs","date":"2024-10-06","arxiv_id":"2410.10849","repositories_listed":0,"syntology":null},{"url":null,"slug":"geometry-is-all-you-need-a-unified-taxonomy","title":"Geometry is All You Need: A Unified Taxonomy of Matrix and Tensor Factorization for Compression of Generative Language Models","date":"2024-10-03","arxiv_id":"2410.03040","repositories_listed":0,"syntology":null},{"url":null,"slug":"compressing-recurrent-neural-networks-for","title":"Compressing Recurrent Neural Networks for FPGA-accelerated Implementation in Fluorescence Lifetime Imaging","date":"2024-10-01","arxiv_id":"2410.00948","repositories_listed":0,"syntology":null},{"url":null,"slug":"aggressive-post-training-compression-on","title":"Aggressive Post-Training Compression on Extremely Large Language Models","date":"2024-09-30","arxiv_id":"2409.20094","repositories_listed":0,"syntology":null},{"url":null,"slug":"infantcrynet-a-data-driven-framework-for","title":"InfantCryNet: A Data-driven Framework for Intelligent Analysis of Infant Cries","date":"2024-09-29","arxiv_id":"2409.19689","repositories_listed":0,"syntology":null},{"url":null,"slug":"value-based-deep-multi-agent-reinforcement","title":"Value-Based Deep Multi-Agent Reinforcement Learning with Dynamic Sparse Training","date":"2024-09-28","arxiv_id":"2409.19391","repositories_listed":0,"syntology":null},{"url":null,"slug":"general-compression-framework-for-efficient","title":"General Compression Framework for Efficient Transformer Object Tracking","date":"2024-09-26","arxiv_id":"2409.17564","repositories_listed":0,"syntology":null},{"url":null,"slug":"applications-of-knowledge-distillation-in","title":"Applications of Knowledge Distillation in Remote Sensing: A Survey","date":"2024-09-18","arxiv_id":"2409.12111","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-preserving-sam-quantization-for","title":"Privacy-Preserving SAM Quantization for Efficient Edge Intelligence in Healthcare","date":"2024-09-14","arxiv_id":"2410.01813","repositories_listed":0,"syntology":null}],"record_sha256":"c11c68fa9985491b79d5f06f699077e1a40f5e8e8a11f9c064400c87f7af3779","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}