{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/quantization/papers/17","list_of":"/task/quantization","task":"Quantization","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":17,"pages_in_order":50,"rows_per_page":100,"rows":[1601,1700],"of":4925,"counts":{"archive_papers_tagged":4925,"with_a_code_link":1596,"where_syntology_ran_a_sample":515,"not_listed_spam_title":0,"listed":4925,"listed_where_code_ran":515,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":452,"every_run_a_failure_of_syntologys_instrument":63,"listed_with_a_run_with_no_instrument_failure":452,"listed_every_run_a_failure_of_syntologys_instrument":63,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/quantization","prev":"/task/quantization/papers/16","next":"/task/quantization/papers/18","papers":[{"url":null,"slug":"lightweight-federated-learning-over-wireless","title":"Lightweight Federated Learning over Wireless Edge Networks","date":"2025-07-13","arxiv_id":"2507.09546","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-foundation-models-as-effective-visual","title":"Vision Foundation Models as Effective Visual Tokenizers for Autoregressive Image Generation","date":"2025-07-11","arxiv_id":"2507.08441","repositories_listed":0,"syntology":null},{"url":null,"slug":"gsvr-2d-gaussian-based-video-representation","title":"GSVR: 2D Gaussian-based Video Representation for 800+ FPS with Hybrid Deformation Field","date":"2025-07-08","arxiv_id":"2507.05594","repositories_listed":0,"syntology":null},{"url":null,"slug":"qs4d-quantization-aware-training-for","title":"QS4D: Quantization-aware training for efficient hardware deployment of structured state-space sequential models","date":"2025-07-08","arxiv_id":"2507.06079","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-certainty-assessment-in-vector","title":"Semantic Certainty Assessment in Vector Retrieval Systems: A Novel Framework for Embedding Quality Evaluation","date":"2025-07-08","arxiv_id":"2507.05933","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-discrete-tokens-treating-them-as","title":"Rethinking Discrete Tokens: Treating Them as Conditions for Continuous Autoregressive Image Synthesis","date":"2025-07-02","arxiv_id":"2507.01756","repositories_listed":0,"syntology":null},{"url":null,"slug":"analysis-of-null-related-beampattern-measures","title":"Analysis of Null Related Beampattern Measures and Signal Quantization Effects for Linear Differential Microphone Arrays","date":"2025-06-26","arxiv_id":"2506.21043","repositories_listed":0,"syntology":null},{"url":null,"slug":"dipsvd-dual-importance-protected-svd-for","title":"DipSVD: Dual-importance Protected SVD for Efficient LLM Compression","date":"2025-06-25","arxiv_id":"2506.20353","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-quantization-and-pruning-neural","title":"Joint Quantization and Pruning Neural Networks Approach: A Case Study on FSO Receivers","date":"2025-06-25","arxiv_id":"2506.20084","repositories_listed":0,"syntology":null},{"url":null,"slug":"cross-layer-discrete-concept-discovery-for","title":"Cross-Layer Discrete Concept Discovery for Interpreting Language Models","date":"2025-06-24","arxiv_id":"2506.20040","repositories_listed":0,"syntology":null},{"url":null,"slug":"variational-bayesian-channel-estimation-and","title":"Variational Bayesian Channel Estimation and Data Detection for Cell-Free Massive MIMO with Low-Resolution Quantized Fronthaul Links","date":"2025-06-23","arxiv_id":"2506.18863","repositories_listed":0,"syntology":null},{"url":null,"slug":"lvpnet-a-latent-variable-based-prediction","title":"LVPNet: A Latent-variable-based Prediction-driven End-to-end Framework for Lossless Compression of Medical Images","date":"2025-06-22","arxiv_id":"2506.17983","repositories_listed":0,"syntology":null},{"url":null,"slug":"stainpidr-a-pathological-image-decouplingand","title":"StainPIDR: A Pathological Image Decouplingand Reconstruction Method for Stain Normalization Based on Color Vector Quantization and Structure Restaining","date":"2025-06-22","arxiv_id":"2506.17879","repositories_listed":0,"syntology":null},{"url":null,"slug":"trojan-guard-hardware-trojans-detection-using","title":"TROJAN-GUARD: Hardware Trojans Detection Using GNN in RTL Designs","date":"2025-06-22","arxiv_id":"2506.17894","repositories_listed":0,"syntology":null},{"url":null,"slug":"rlrc-reinforcement-learning-based-recovery","title":"RLRC: Reinforcement Learning-based Recovery for Compressed Vision-Language-Action Models","date":"2025-06-21","arxiv_id":"2506.17639","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-simple-contrastive-framework-of-item","title":"A Simple Contrastive Framework Of Item Tokenization For Generative Recommendation","date":"2025-06-20","arxiv_id":"2506.16683","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-hidden-cost-of-an-image-quantifying-the","title":"The Hidden Cost of an Image: Quantifying the Energy Consumption of AI Image Generation","date":"2025-06-20","arxiv_id":"2506.17016","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-designing-modulation-for-over-the-air-1","title":"On Designing Modulation for Over-the-Air Computation -- Part I: Noise-Aware Design","date":"2025-06-19","arxiv_id":"2506.15950","repositories_listed":0,"syntology":null},{"url":null,"slug":"paroattention-pattern-aware-reordering-for","title":"PAROAttention: Pattern-Aware ReOrdering for Efficient Sparse and Quantized Attention in Visual Generation Models","date":"2025-06-19","arxiv_id":"2506.16054","repositories_listed":0,"syntology":null},{"url":null,"slug":"effect-of-signal-quantization-on-performance","title":"Effect of Signal Quantization on Performance Measures of a 1st Order One Dimensional Differential Microphone Array","date":"2025-06-18","arxiv_id":"2506.15463","repositories_listed":0,"syntology":null},{"url":null,"slug":"j3dai-a-tiny-dnn-based-edge-ai-accelerator","title":"J3DAI: A tiny DNN-Based Edge AI Accelerator for 3D-Stacked CMOS Image Sensor","date":"2025-06-18","arxiv_id":"2506.15316","repositories_listed":0,"syntology":null},{"url":null,"slug":"compressed-video-super-resolution-based-on","title":"Compressed Video Super-Resolution based on Hierarchical Encoding","date":"2025-06-17","arxiv_id":"2506.14381","repositories_listed":0,"syntology":null},{"url":null,"slug":"cost-aware-routing-for-efficient-text-to","title":"Cost-Aware Routing for Efficient Text-To-Image Generation","date":"2025-06-17","arxiv_id":"2506.14753","repositories_listed":0,"syntology":null},{"url":null,"slug":"mote-mixture-of-ternary-experts-for-memory","title":"MoTE: Mixture of Ternary Experts for Memory-efficient Large Multimodal Models","date":"2025-06-17","arxiv_id":"2506.14435","repositories_listed":0,"syntology":null},{"url":null,"slug":"eaquant-enhancing-post-training-quantization","title":"EAQuant: Enhancing Post-Training Quantization for MoE Models via Expert-Aware Optimization","date":"2025-06-16","arxiv_id":"2506.13329","repositories_listed":0,"syntology":null},{"url":null,"slug":"rosaq-rotation-based-saliency-aware-weight","title":"ROSAQ: Rotation-based Saliency-Aware Weight Quantization for Efficiently Compressing Large Language Models","date":"2025-06-16","arxiv_id":"2506.13472","repositories_listed":0,"syntology":null},{"url":null,"slug":"serving-large-language-models-on-huawei","title":"Serving Large Language Models on Huawei CloudMatrix384","date":"2025-06-15","arxiv_id":"2506.12708","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantizing-small-scale-state-space-models-for","title":"Quantizing Small-Scale State-Space Models for Edge AI","date":"2025-06-14","arxiv_id":"2506.12480","repositories_listed":0,"syntology":null},{"url":null,"slug":"relative-entropy-regularized-reinforcement","title":"Relative Entropy Regularized Reinforcement Learning for Efficient Encrypted Policy Synthesis","date":"2025-06-14","arxiv_id":"2506.12358","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-model-acceleration-and","title":"Deep Learning Model Acceleration and Optimization Strategies for Real-Time Recommendation Systems","date":"2025-06-13","arxiv_id":"2506.11421","repositories_listed":0,"syntology":null},{"url":null,"slug":"gplq-a-general-practical-and-lightning-qat","title":"GPLQ: A General, Practical, and Lightning QAT Method for Vision Transformers","date":"2025-06-13","arxiv_id":"2506.11784","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-10274","title":"Discrete Audio Tokens: More Than a Survey!","date":"2025-06-12","arxiv_id":"2506.10274","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-10463","title":"Starting Positions Matter: A Study on Better Weight Initialization for Neural Network Quantization","date":"2025-06-12","arxiv_id":"2506.10463","repositories_listed":0,"syntology":null},{"url":null,"slug":"mnn-llm-a-generic-inference-engine-for-fast","title":"MNN-LLM: A Generic Inference Engine for Fast Large Language Model Deployment on Mobile Devices","date":"2025-06-12","arxiv_id":"2506.10443","repositories_listed":0,"syntology":null},{"url":null,"slug":"post-training-quantization-for-video-matting","title":"Post-Training Quantization for Video Matting","date":"2025-06-12","arxiv_id":"2506.10840","repositories_listed":0,"syntology":null},{"url":null,"slug":"awp-activation-aware-weight-pruning-and","title":"AWP: Activation-Aware Weight Pruning and Quantization with Projected Gradient Descent","date":"2025-06-11","arxiv_id":"2506.10205","repositories_listed":0,"syntology":null},{"url":null,"slug":"hadanorm-diffusion-transformer-quantization","title":"HadaNorm: Diffusion Transformer Quantization through Mean-Centered Transformations","date":"2025-06-11","arxiv_id":"2506.09932","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-sam2-accurate-quantization-for-segment","title":"Q-SAM2: Accurate Quantization for Segment Anything Model 2","date":"2025-06-11","arxiv_id":"2506.09782","repositories_listed":0,"syntology":null},{"url":null,"slug":"sled-a-speculative-llm-decoding-framework-for","title":"SLED: A Speculative LLM Decoding Framework for Efficient Edge Serving","date":"2025-06-11","arxiv_id":"2506.09397","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08662","title":"Optimizing Learned Image Compression on Scalar and Entropy-Constraint Quantization","date":"2025-06-10","arxiv_id":"2506.08662","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08785","title":"POLARON: Precision-aware On-device Learning and Adaptive Runtime-cONfigurable AI acceleration","date":"2025-06-10","arxiv_id":"2506.08785","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-08911","title":"Implementing Keyword Spotting on the MCUX947 Microcontroller with Integrated NPU","date":"2025-06-10","arxiv_id":"2506.08911","repositories_listed":0,"syntology":null},{"url":null,"slug":"hardware-limitations-and-optimization","title":"Hardware Limitations and Optimization Approach in 1-Bit RIS Design at 28 GHz","date":"2025-06-10","arxiv_id":"2506.08930","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-optimization-on-compact","title":"Decentralized Optimization on Compact Submanifolds by Quantized Riemannian Gradient Tracking","date":"2025-06-09","arxiv_id":"2506.07351","repositories_listed":0,"syntology":null},{"url":null,"slug":"litevlm-a-low-latency-vision-language-model","title":"LiteVLM: A Low-Latency Vision-Language Model Inference Pipeline for Resource-Constrained Environments","date":"2025-06-09","arxiv_id":"2506.07416","repositories_listed":0,"syntology":null},{"url":null,"slug":"2506-06975","title":"Auditing Black-Box LLM APIs with a Rank-Based Uniformity Test","date":"2025-06-08","arxiv_id":"2506.06975","repositories_listed":0,"syntology":null},{"url":null,"slug":"qforce-rl-quantized-fpga-optimized","title":"QForce-RL: Quantized FPGA-Optimized Reinforcement Learning Compute Engine","date":"2025-06-08","arxiv_id":"2506.07046","repositories_listed":0,"syntology":null},{"url":null,"slug":"enabling-on-device-medical-ai-assistants-via","title":"Enabling On-Device Medical AI Assistants via Input-Driven Saliency Adaptation","date":"2025-06-07","arxiv_id":"2506.11105","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-ai-native-fronthaul-neural","title":"Towards AI-Native Fronthaul: Neural Compression for NextG Cloud RAN","date":"2025-06-07","arxiv_id":"2506.06925","repositories_listed":0,"syntology":null},{"url":null,"slug":"beast-efficient-tokenization-of-b-splines","title":"BEAST: Efficient Tokenization of B-Splines Encoded Action Sequences for Imitation Learning","date":"2025-06-06","arxiv_id":"2506.06072","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-modality-gap-softly-discretizing","title":"Bridging the Modality Gap: Softly Discretizing Audio Representation for LLM-based Automatic Speech Recognition","date":"2025-06-06","arxiv_id":"2506.05706","repositories_listed":0,"syntology":null},{"url":null,"slug":"fpsattention-training-aware-fp8-and-sparsity","title":"FPSAttention: Training-Aware FP8 and Sparsity Co-Design for Fast Video Diffusion","date":"2025-06-05","arxiv_id":"2506.04648","repositories_listed":0,"syntology":null},{"url":null,"slug":"fptquant-function-preserving-transforms-for","title":"FPTQuant: Function-Preserving Transforms for LLM Quantization","date":"2025-06-05","arxiv_id":"2506.04985","repositories_listed":0,"syntology":null},{"url":null,"slug":"kernel-k-medoids-as-general-vector","title":"Kernel $k$-Medoids as General Vector Quantization","date":"2025-06-05","arxiv_id":"2506.04786","repositories_listed":0,"syntology":null},{"url":null,"slug":"massive-mimo-with-1-bit-dacs-data-detection","title":"Massive MIMO with 1-Bit DACs: Data Detection for Quantized Linear Precoding with Dithering","date":"2025-06-05","arxiv_id":"2506.05072","repositories_listed":0,"syntology":null},{"url":null,"slug":"pcdvq-enhancing-vector-quantization-for-large","title":"PCDVQ: Enhancing Vector Quantization for Large Language Models via Polar Coordinate Decoupling","date":"2025-06-05","arxiv_id":"2506.05432","repositories_listed":0,"syntology":null},{"url":null,"slug":"tada-training-free-recipe-for-decoding-with","title":"TaDA: Training-free recipe for Decoding with Adaptive KV Cache Compression and Mean-centering","date":"2025-06-05","arxiv_id":"2506.04642","repositories_listed":0,"syntology":null},{"url":null,"slug":"bittts-highly-compact-text-to-speech-using-1","title":"BitTTS: Highly Compact Text-to-Speech Using 1.58-bit Quantization and Weight Indexing","date":"2025-06-04","arxiv_id":"2506.03515","repositories_listed":0,"syntology":null},{"url":null,"slug":"nonlinear-sparse-bayesian-learning-methods","title":"Nonlinear Sparse Bayesian Learning Methods with Application to Massive MIMO Channel Estimation with Hardware Impairments","date":"2025-06-04","arxiv_id":"2506.03775","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-convergence-privacy-and-fairness","title":"Enhancing Convergence, Privacy and Fairness for Wireless Personalized Federated Learning: Quantization-Assisted Min-Max Fair Scheduling","date":"2025-06-03","arxiv_id":"2506.02422","repositories_listed":0,"syntology":null},{"url":null,"slug":"muc-g4-minimal-unsat-core-guided-incremental","title":"MUC-G4: Minimal Unsat Core-Guided Incremental Verification for Deep Neural Network Compression","date":"2025-06-03","arxiv_id":"2506.04268","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantized-dissipative-uncertain-model-for","title":"Quantized Dissipative Uncertain Model for Fractional T_S Fuzzy systems with Time_Varying Delays Under Networked Control System","date":"2025-06-03","arxiv_id":"2506.02788","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-speech-emotion-recognition-with","title":"Enhancing Speech Emotion Recognition with Graph-Based Multimodal Fusion and Prosodic Features for the Speech Emotion Recognition in Naturalistic Conditions Challenge at Interspeech 2025","date":"2025-06-02","arxiv_id":"2506.02088","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantitative-error-feedback-for-quantization","title":"Quantitative Error Feedback for Quantization Noise Reduction of Filtering over Graphs","date":"2025-06-02","arxiv_id":"2506.01404","repositories_listed":0,"syntology":null},{"url":null,"slug":"clap-art-automated-audio-captioning-with","title":"CLAP-ART: Automated Audio Captioning with Semantic-rich Audio Representation Tokenizer","date":"2025-06-01","arxiv_id":"2506.00800","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantization-based-bounds-on-the-wasserstein","title":"Quantization-based Bounds on the Wasserstein Metric","date":"2025-06-01","arxiv_id":"2506.00976","repositories_listed":0,"syntology":null},{"url":null,"slug":"power-of-two-pot-weights-in-large-language","title":"Power-of-Two (PoT) Weights in Large Language Models (LLMs)","date":"2025-05-31","arxiv_id":"2506.00315","repositories_listed":0,"syntology":null},{"url":null,"slug":"edge-computing-for-physics-driven-ai-in","title":"Edge Computing for Physics-Driven AI in Computational MRI: A Feasibility Study","date":"2025-05-30","arxiv_id":"2506.03183","repositories_listed":0,"syntology":null},{"url":"/paper/littlebit-ultra-low-bit-quantization-via","slug":"littlebit-ultra-low-bit-quantization-via","title":"LittleBit: Ultra Low-Bit Quantization via Latent Factorization","date":"2025-05-30","arxiv_id":"2506.13771","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/littlebit-ultra-low-bit-quantization-via#ran","syntology_url":"https://syntology.ai/paper/2506.13771","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13771"}},"official":null}},{"url":null,"slug":"running-conventional-automatic-speech","title":"Running Conventional Automatic Speech Recognition on Memristor Hardware: A Simulated Approach","date":"2025-05-30","arxiv_id":"2505.24721","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-quantum-approximate-k-nn-algorithm","title":"Efficient Quantum Approximate $k$NN Algorithm via Granular-Ball Computing","date":"2025-05-29","arxiv_id":"2505.23066","repositories_listed":0,"syntology":null},{"url":null,"slug":"muloco-muon-is-a-practical-inner-optimizer","title":"MuLoCo: Muon is a practical inner optimizer for DiLoCo","date":"2025-05-29","arxiv_id":"2505.23725","repositories_listed":0,"syntology":null},{"url":null,"slug":"revisiting-uncertainty-estimation-and","title":"Revisiting Uncertainty Estimation and Calibration of Large Language Models","date":"2025-05-29","arxiv_id":"2505.23854","repositories_listed":0,"syntology":null},{"url":null,"slug":"highly-efficient-and-effective-llms-with","title":"Highly Efficient and Effective LLMs with Multi-Boolean Architectures","date":"2025-05-28","arxiv_id":"2505.22811","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-interplay-of-privacy-persuasion-and","title":"On the Interplay of Privacy, Persuasion and Quantization","date":"2025-05-28","arxiv_id":"2506.06321","repositories_listed":0,"syntology":null},{"url":null,"slug":"brainstratify-coarse-to-fine-disentanglement","title":"BrainStratify: Coarse-to-Fine Disentanglement of Intracranial Neural Dynamics","date":"2025-05-26","arxiv_id":"2505.20480","repositories_listed":0,"syntology":null},{"url":null,"slug":"ca3d-convolutional-attentional-3d-nets-for","title":"CA3D: Convolutional-Attentional 3D Nets for Efficient Video Activity Recognition on the Edge","date":"2025-05-26","arxiv_id":"2505.19928","repositories_listed":0,"syntology":null},{"url":null,"slug":"lpcm-learning-based-predictive-coding-for","title":"LPCM: Learning-based Predictive Coding for LiDAR Point Cloud Compression","date":"2025-05-26","arxiv_id":"2505.20059","repositories_listed":0,"syntology":null},{"url":null,"slug":"small-language-models-architectures","title":"Small Language Models: Architectures, Techniques, Evaluation, Problems and Future Adaptation","date":"2025-05-26","arxiv_id":"2505.19529","repositories_listed":0,"syntology":null},{"url":null,"slug":"fastmamba-a-high-speed-and-efficient-mamba","title":"FastMamba: A High-Speed and Efficient Mamba Accelerator on FPGA with Accurate Quantization","date":"2025-05-25","arxiv_id":"2505.18975","repositories_listed":0,"syntology":null},{"url":null,"slug":"distinctive-feature-codec-adaptive","title":"Distinctive Feature Codec: Adaptive Segmentation for Efficient Speech Representation","date":"2025-05-24","arxiv_id":"2505.18516","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-and-workload-aware-llm-serving-via","title":"Efficient and Workload-Aware LLM Serving via Runtime Layer Swapping and KV Cache Resizing","date":"2025-05-24","arxiv_id":"2506.02006","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-discreteness-finite-sample-analysis-of","title":"Beyond Discreteness: Finite-Sample Analysis of Straight-Through Estimator for Quantization","date":"2025-05-23","arxiv_id":"2505.18113","repositories_listed":0,"syntology":null},{"url":null,"slug":"nsnquant-a-double-normalization-approach-for","title":"NSNQuant: A Double Normalization Approach for Calibration-Free Low-Bit Vector Quantization of KV Cache","date":"2025-05-23","arxiv_id":"2505.18231","repositories_listed":0,"syntology":null},{"url":null,"slug":"slot-mllm-object-centric-visual-tokenization","title":"Slot-MLLM: Object-Centric Visual Tokenization for Multimodal LLM","date":"2025-05-23","arxiv_id":"2505.17726","repositories_listed":0,"syntology":null},{"url":null,"slug":"task-specific-pruning-with-llm-sieve-how-many","title":"Task Specific Pruning with LLM-Sieve: How Many Parameters Does Your Task Really Need?","date":"2025-05-23","arxiv_id":"2505.18350","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-quantum-optimization-ready-an-effort","title":"Is Quantum Optimization Ready? An Effort Towards Neural Network Compression using Adiabatic Quantum Computing","date":"2025-05-22","arxiv_id":"2505.16332","repositories_listed":0,"syntology":null},{"url":null,"slug":"nqkv-a-kv-cache-quantization-scheme-based-on","title":"NQKV: A KV Cache Quantization Scheme Based on Normal Distribution Characteristics","date":"2025-05-22","arxiv_id":"2505.16210","repositories_listed":0,"syntology":null},{"url":null,"slug":"harnessing-large-language-models-locally","title":"Harnessing Large Language Models Locally: Empirical Results and Implications for AI PC","date":"2025-05-21","arxiv_id":"2505.15030","repositories_listed":0,"syntology":null},{"url":null,"slug":"intreeger-an-end-to-end-framework-for-integer","title":"InTreeger: An End-to-End Framework for Integer-Only Decision Tree Inference","date":"2025-05-21","arxiv_id":"2505.15391","repositories_listed":0,"syntology":null},{"url":null,"slug":"is-selective-round-to-nearest-quantization","title":"Is (Selective) Round-To-Nearest Quantization All You Need?","date":"2025-05-21","arxiv_id":"2505.15909","repositories_listed":0,"syntology":null},{"url":null,"slug":"rate-distortion-optimization-with-non","title":"Rate-Distortion Optimization with Non-Reference Metrics for UGC Compression","date":"2025-05-21","arxiv_id":"2505.15003","repositories_listed":0,"syntology":null},{"url":null,"slug":"segmentation-variant-codebooks-for","title":"Segmentation-Variant Codebooks for Preservation of Paralinguistic and Prosodic Information","date":"2025-05-21","arxiv_id":"2505.15667","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficientllm-efficiency-in-large-language","title":"EfficientLLM: Efficiency in Large Language Models","date":"2025-05-20","arxiv_id":"2505.13840","repositories_listed":0,"syntology":null},{"url":null,"slug":"layer-wise-quantization-for-quantized","title":"Layer-wise Quantization for Quantized Optimistic Dual Averaging","date":"2025-05-20","arxiv_id":"2505.14371","repositories_listed":0,"syntology":null},{"url":null,"slug":"through-a-compressed-lens-investigating-the","title":"Through a Compressed Lens: Investigating the Impact of Quantization on LLM Explainability and Interpretability","date":"2025-05-20","arxiv_id":"2505.13963","repositories_listed":0,"syntology":null},{"url":null,"slug":"a3-an-analytical-low-rank-approximation","title":"A3 : an Analytical Low-Rank Approximation Framework for Attention","date":"2025-05-19","arxiv_id":"2505.12942","repositories_listed":0,"syntology":null},{"url":null,"slug":"automatic-mixed-precision-for-optimizing","title":"Automatic mixed precision for optimizing gained time with constrained loss mean-squared-error based on model partition to sequential sub-graphs","date":"2025-05-19","arxiv_id":"2505.13060","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-unfolding-with-kernel-based-quantization","title":"Deep Unfolding with Kernel-based Quantization in MIMO Detection","date":"2025-05-19","arxiv_id":"2505.12736","repositories_listed":0,"syntology":null},{"url":null,"slug":"gancompress-gan-enhanced-neural-image","title":"GANCompress: GAN-Enhanced Neural Image Compression with Binary Spherical Quantization","date":"2025-05-19","arxiv_id":"2505.13542","repositories_listed":0,"syntology":null}],"record_sha256":"3a188847ad693a4d46d4a80d5ea00beb57e604021fd4c842567489fbbbab1007","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}