{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/quantization/papers/20","list_of":"/task/quantization","task":"Quantization","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":20,"pages_in_order":50,"rows_per_page":100,"rows":[1901,2000],"of":4925,"counts":{"archive_papers_tagged":4925,"with_a_code_link":1596,"where_syntology_ran_a_sample":515,"not_listed_spam_title":0,"listed":4925,"listed_where_code_ran":515,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":452,"every_run_a_failure_of_syntologys_instrument":63,"listed_with_a_run_with_no_instrument_failure":452,"listed_every_run_a_failure_of_syntologys_instrument":63,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/quantization","prev":"/task/quantization/papers/19","next":"/task/quantization/papers/21","papers":[{"url":null,"slug":"verification-of-bit-flip-attacks-against","title":"Verification of Bit-Flip Attacks against Quantized Neural Networks","date":"2025-02-22","arxiv_id":"2502.16286","repositories_listed":0,"syntology":null},{"url":null,"slug":"exact-recovery-of-sparse-binary-vectors-from","title":"Exact Recovery of Sparse Binary Vectors from Generalized Linear Measurements","date":"2025-02-21","arxiv_id":"2502.16008","repositories_listed":0,"syntology":null},{"url":null,"slug":"fd-lscic-frequency-decomposition-based","title":"FD-LSCIC: Frequency Decomposition-based Learned Screen Content Image Compression","date":"2025-02-21","arxiv_id":"2502.15174","repositories_listed":0,"syntology":null},{"url":null,"slug":"interleaved-block-based-learned-image","title":"Interleaved Block-based Learned Image Compression with Feature Enhancement and Quantization Error Compensation","date":"2025-02-21","arxiv_id":"2502.15188","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-petr-quant-aware-position-embedding","title":"Q-PETR: Quant-aware Position Embedding Transformation for Multi-View 3D Object Detection","date":"2025-02-21","arxiv_id":"2502.15488","repositories_listed":0,"syntology":null},{"url":null,"slug":"svdq-1-25-bit-and-410x-key-cache-compression","title":"SVDq: 1.25-bit and 410x Key Cache Compression for LLM Attention","date":"2025-02-21","arxiv_id":"2502.15304","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-compression-meets-model-compression","title":"When Compression Meets Model Compression: Memory-Efficient Double Compression for Large Language Models","date":"2025-02-21","arxiv_id":"2502.15443","repositories_listed":0,"syntology":null},{"url":"/paper/efficient-ai-in-practice-training-and","slug":"efficient-ai-in-practice-training-and","title":"Efficient AI in Practice: Training and Deployment of Efficient LLMs for Industry Applications","date":"2025-02-20","arxiv_id":"2502.14305","repositories_listed":0,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/efficient-ai-in-practice-training-and#ran","syntology_url":"https://syntology.ai/paper/2502.14305","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.14305"}},"official":null}},{"url":null,"slug":"hardware-friendly-static-quantization-method","title":"Hardware-Friendly Static Quantization Method for Video Diffusion Transformers","date":"2025-02-20","arxiv_id":"2502.15077","repositories_listed":0,"syntology":null},{"url":null,"slug":"more-for-keys-less-for-values-adaptive-kv","title":"More for Keys, Less for Values: Adaptive KV Cache Quantization","date":"2025-02-20","arxiv_id":"2502.15075","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-general-error-theoretical-analysis","title":"A General Error-Theoretical Analysis Framework for Constructing Compression Strategies","date":"2025-02-19","arxiv_id":"2502.15802","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-2-ats-retrieval-based-kv-cache-reduction","title":"A$^2$ATS: Retrieval-Based KV Cache Reduction via Windowed Rotary Position Embedding and Query-Aware Vector Quantization","date":"2025-02-18","arxiv_id":"2502.12665","repositories_listed":0,"syntology":null},{"url":null,"slug":"continual-quantization-aware-pre-training","title":"Continual Quantization-Aware Pre-Training: When to transition from 16-bit to 1.58-bit pre-training for BitNet language models?","date":"2025-02-17","arxiv_id":"2502.11895","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-logic-elements-associated-with-round","title":"On the Logic Elements Associated with Round-Off Errors and Gaussian Blur in Image Registration: A Simple Case of Commingling","date":"2025-02-17","arxiv_id":"2502.11992","repositories_listed":0,"syntology":null},{"url":null,"slug":"rotate-clip-and-partition-towards-w2a4kv4","title":"Rotate, Clip, and Partition: Towards W2A4KV4 Quantization by Integrating Rotation and Learnable Non-uniform Quantizer","date":"2025-02-17","arxiv_id":"2502.15779","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-efficient-pre-training-exploring-fp4","title":"Towards Efficient Pre-training: Exploring FP4 Precision in Large Language Models","date":"2025-02-17","arxiv_id":"2502.11458","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-reasoning-ability-of-small-language","title":"Towards Reasoning Ability of Small Language Models","date":"2025-02-17","arxiv_id":"2502.11569","repositories_listed":0,"syntology":null},{"url":null,"slug":"embbert-q-breaking-memory-barriers-in","title":"EmbBERT-Q: Breaking Memory Barriers in Embedded NLP","date":"2025-02-14","arxiv_id":"2502.10001","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-complexity-on-grid-channel-estimation-for","title":"Low-Complexity On-Grid Channel Estimation for Partially-Connected Hybrid XL-MIMO","date":"2025-02-14","arxiv_id":"2502.09929","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-watermarking-of-open-source-llms","title":"Towards Watermarking of Open-Source LLMs","date":"2025-02-14","arxiv_id":"2502.10525","repositories_listed":0,"syntology":null},{"url":null,"slug":"nestquant-nested-lattice-quantization-for","title":"NestQuant: Nested Lattice Quantization for Matrix Products and LLMs","date":"2025-02-13","arxiv_id":"2502.09720","repositories_listed":0,"syntology":null},{"url":"/paper/roste-an-efficient-quantization-aware","slug":"roste-an-efficient-quantization-aware","title":"RoSTE: An Efficient Quantization-Aware Supervised Fine-Tuning Approach for Large Language Models","date":"2025-02-13","arxiv_id":"2502.09003","repositories_listed":0,"syntology":{"n":5,"n_ran":5,"n_constructed":5,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 5 samples that ran constructed an object rather than computing a result","sample_list":"/paper/roste-an-efficient-quantization-aware#ran","syntology_url":"https://syntology.ai/paper/2502.09003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.09003"}},"official":null}},{"url":null,"slug":"compression-of-site-specific-deep-neural","title":"Compression of Site-Specific Deep Neural Networks for Massive MIMO Precoding","date":"2025-02-12","arxiv_id":"2502.08758","repositories_listed":0,"syntology":null},{"url":null,"slug":"contextual-compression-encoding-for-large","title":"Contextual Compression Encoding for Large Language Models: A Novel Framework for Multi-Layered Parameter Space Pruning","date":"2025-02-12","arxiv_id":"2502.08323","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploiting-non-uniform-quantization-for","title":"Exploiting Non-uniform Quantization for Enhanced ILC in Wideband Digital Pre-distortion","date":"2025-02-12","arxiv_id":"2502.08360","repositories_listed":0,"syntology":null},{"url":null,"slug":"lowra-accurate-and-efficient-lora-fine-tuning","title":"LowRA: Accurate and Efficient LoRA Fine-Tuning of LLMs under 2 Bits","date":"2025-02-12","arxiv_id":"2502.08141","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-thermodynamic-second-order","title":"Scalable Thermodynamic Second-order Optimization","date":"2025-02-12","arxiv_id":"2502.08603","repositories_listed":0,"syntology":null},{"url":null,"slug":"conditional-distribution-quantization-in","title":"Conditional Distribution Quantization in Machine Learning","date":"2025-02-11","arxiv_id":"2502.07151","repositories_listed":0,"syntology":null},{"url":null,"slug":"hdcompression-hybrid-diffusion-image","title":"HDCompression: Hybrid-Diffusion Image Compression for Ultra-Low Bitrates","date":"2025-02-11","arxiv_id":"2502.07160","repositories_listed":0,"syntology":null},{"url":null,"slug":"memhd-memory-efficient-multi-centroid","title":"MEMHD: Memory-Efficient Multi-Centroid Hyperdimensional Computing for Fully-Utilized In-Memory Computing Architectures","date":"2025-02-11","arxiv_id":"2502.07834","repositories_listed":0,"syntology":null},{"url":null,"slug":"vision-language-models-for-edge-networks-a","title":"Vision-Language Models for Edge Networks: A Comprehensive Survey","date":"2025-02-11","arxiv_id":"2502.07855","repositories_listed":0,"syntology":null},{"url":null,"slug":"demystifying-singular-defects-in-large","title":"Demystifying Singular Defects in Large Language Models","date":"2025-02-10","arxiv_id":"2502.07004","repositories_listed":0,"syntology":null},{"url":null,"slug":"finetuning-and-quantization-of-eeg-based","title":"Finetuning and Quantization of EEG-Based Foundational BioSignal Models on ECG and PPG Data for Blood Pressure Estimation","date":"2025-02-10","arxiv_id":"2502.17460","repositories_listed":0,"syntology":null},{"url":null,"slug":"matryoshka-quantization","title":"Matryoshka Quantization","date":"2025-02-10","arxiv_id":"2502.06786","repositories_listed":0,"syntology":null},{"url":null,"slug":"gradient-based-method-for-the-fusion-of","title":"Gradient Based Method for the Fusion of Lattice Quantizers","date":"2025-02-09","arxiv_id":"2502.06887","repositories_listed":0,"syntology":null},{"url":null,"slug":"aiqvit-architecture-informed-post-training","title":"AIQViT: Architecture-Informed Post-Training Quantization for Vision Transformers","date":"2025-02-07","arxiv_id":"2502.04628","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-evaluation-of-quantization-effects","title":"Efficient Evaluation of Quantization-Effects in Neural Codecs","date":"2025-02-07","arxiv_id":"2502.04770","repositories_listed":0,"syntology":null},{"url":null,"slug":"qlip-text-aligned-visual-tokenization-unifies","title":"QLIP: Text-Aligned Visual Tokenization Unifies Auto-Regressive Multimodal Understanding and Generation","date":"2025-02-07","arxiv_id":"2502.05178","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-and-consistent-embedding-of","title":"Scalable and consistent embedding of probability measures into Hilbert spaces via measure quantization","date":"2025-02-07","arxiv_id":"2502.04907","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-performance-analysis-of-you-only-look-once","title":"A Performance Analysis of You Only Look Once Models for Deployment on Constrained Computational Edge Devices in Drone Applications","date":"2025-02-06","arxiv_id":"2502.15737","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-model-invariance-with-discrete","title":"Exploring Model Invariance with Discrete Search for Ultra-Low-Bit Quantization","date":"2025-02-06","arxiv_id":"2502.06844","repositories_listed":0,"syntology":null},{"url":null,"slug":"tq-dit-efficient-time-aware-quantization-for","title":"TQ-DiT: Efficient Time-Aware Quantization for Diffusion Transformers","date":"2025-02-06","arxiv_id":"2502.04056","repositories_listed":0,"syntology":null},{"url":null,"slug":"asymptotic-analysis-of-one-bit-quantized-box","title":"Asymptotic Analysis of One-bit Quantized Box-Constrained Precoding in Large-Scale Multi-User Systems","date":"2025-02-05","arxiv_id":"2502.02953","repositories_listed":0,"syntology":null},{"url":null,"slug":"hack-homomorphic-acceleration-via-compression","title":"HACK: Homomorphic Acceleration via Compression of the Key-Value Cache for Disaggregated LLM Inference","date":"2025-02-05","arxiv_id":"2502.03589","repositories_listed":0,"syntology":null},{"url":null,"slug":"sensorchat-answering-qualitative-and","title":"SensorChat: Answering Qualitative and Quantitative Questions during Long-Term Multimodal Sensor Interactions","date":"2025-02-05","arxiv_id":"2502.02883","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-of-quantization-techniques-for-on","title":"Survey of Quantization Techniques for On-Device Vision-based Crack Detection","date":"2025-02-04","arxiv_id":"2502.02269","repositories_listed":0,"syntology":null},{"url":null,"slug":"unlocking-efficient-large-inference-models","title":"Unlocking Efficient Large Inference Models: One-Bit Unrolling Tips the Scales","date":"2025-02-04","arxiv_id":"2502.01908","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-investigation-of-fp8-across-accelerators","title":"An Inquiry into Datacenter TCO for LLM Inference with FP8","date":"2025-02-03","arxiv_id":"2502.01070","repositories_listed":0,"syntology":null},{"url":null,"slug":"choose-your-model-size-any-compression-by-a","title":"Choose Your Model Size: Any Compression by a Single Gradient Descent","date":"2025-02-03","arxiv_id":"2502.01717","repositories_listed":0,"syntology":null},{"url":null,"slug":"continuous-autoregressive-modeling-with","title":"Continuous Autoregressive Modeling with Stochastic Monotonic Alignment for Speech Synthesis","date":"2025-02-03","arxiv_id":"2502.01084","repositories_listed":0,"syntology":null},{"url":null,"slug":"huff-llm-end-to-end-lossless-compression-for","title":"Huff-LLM: End-to-End Lossless Compression for Efficient LLM Inference","date":"2025-02-02","arxiv_id":"2502.00922","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-noncommutative-quantum-mechanics-and-the","title":"On Noncommutative Quantum Mechanics and the Black-Scholes Model","date":"2025-02-02","arxiv_id":"2502.00938","repositories_listed":0,"syntology":null},{"url":null,"slug":"structural-latency-perturbation-in-large","title":"Structural Latency Perturbation in Large Language Models Through Recursive State Induction","date":"2025-02-02","arxiv_id":"2502.00758","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-field-oriented-control-of-electric","title":"Enhancing Field-Oriented Control of Electric Drives with Tiny Neural Network Optimized for Micro-controllers","date":"2025-02-01","arxiv_id":"2502.00532","repositories_listed":0,"syntology":null},{"url":null,"slug":"mquant-unleashing-the-inference-potential-of","title":"MQuant: Unleashing the Inference Potential of Multimodal Large Language Models via Full Static Quantization","date":"2025-02-01","arxiv_id":"2502.00425","repositories_listed":0,"syntology":null},{"url":null,"slug":"fully-distributed-and-quantized-algorithm-for","title":"Fully Distributed and Quantized Algorithm for MPC-based Autonomous Vehicle Platooning Optimization","date":"2025-01-31","arxiv_id":"2501.18889","repositories_listed":0,"syntology":null},{"url":null,"slug":"llm-based-affective-text-generation-quality","title":"LLM-based Affective Text Generation Quality Based on Different Quantization Values","date":"2025-01-31","arxiv_id":"2501.19317","repositories_listed":0,"syntology":null},{"url":null,"slug":"codebrain-impute-any-brain-mri-via-instance","title":"CodeBrain: Impute Any Brain MRI via Instance-specific Scalar-quantized Codes","date":"2025-01-30","arxiv_id":"2501.18328","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixed-precision-graph-neural-quantization-for","title":"Mixed-Precision Graph Neural Quantization for Low Bit Large Language Models","date":"2025-01-30","arxiv_id":"2501.18154","repositories_listed":0,"syntology":null},{"url":null,"slug":"distinguished-quantized-guidance-for","title":"Distinguished Quantized Guidance for Diffusion-based Sequence Recommendation","date":"2025-01-29","arxiv_id":"2501.17670","repositories_listed":0,"syntology":null},{"url":null,"slug":"edgemlops-operationalizing-ml-models-with","title":"EdgeMLOps: Operationalizing ML models with Cumulocity IoT and thin-edge.io for Visual quality Inspection","date":"2025-01-28","arxiv_id":"2501.17062","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-large-language-model-training","title":"Optimizing Large Language Model Training Using FP4 Quantization","date":"2025-01-28","arxiv_id":"2501.17116","repositories_listed":0,"syntology":null},{"url":null,"slug":"post-training-quantization-for-3d-medical","title":"Post-Training Quantization for 3D Medical Image Segmentation: A Practical Study on Real Inference Engines","date":"2025-01-28","arxiv_id":"2501.17343","repositories_listed":0,"syntology":null},{"url":null,"slug":"post-training-quantization-for-vision-mamba","title":"Post-Training Quantization for Vision Mamba with k-Scaled Quantization and Reparameterization","date":"2025-01-28","arxiv_id":"2501.16738","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-bit-sigma-delta-dfrc-waveform-design","title":"One-Bit Sigma-Delta DFRC Waveform Design: Using Quantization Noise for Radar Probing","date":"2025-01-27","arxiv_id":"2501.15868","repositories_listed":0,"syntology":null},{"url":null,"slug":"stabilization-of-an-unstable-reaction","title":"Stabilization of an unstable reaction-diffusion PDE with input delay despite state and input quantization","date":"2025-01-27","arxiv_id":"2501.15924","repositories_listed":0,"syntology":null},{"url":null,"slug":"decentralized-low-rank-fine-tuning-of-large","title":"Decentralized Low-Rank Fine-Tuning of Large Language Models","date":"2025-01-26","arxiv_id":"2501.15361","repositories_listed":0,"syntology":null},{"url":null,"slug":"sq-dm-accelerating-diffusion-models-with","title":"SQ-DM: Accelerating Diffusion Models with Aggressive Quantization and Temporal Sparsity","date":"2025-01-26","arxiv_id":"2501.15448","repositories_listed":0,"syntology":null},{"url":null,"slug":"akvq-vl-attention-aware-kv-cache-adaptive-2","title":"AKVQ-VL: Attention-Aware KV Cache Adaptive 2-Bit Quantization for Vision-Language Models","date":"2025-01-25","arxiv_id":"2501.15021","repositories_listed":0,"syntology":null},{"url":null,"slug":"fbquant-feedback-quantization-for-large","title":"FBQuant: FeedBack Quantization for Large Language Models","date":"2025-01-25","arxiv_id":"2501.16385","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-accelerating-edge-ai-optimizing-resource","title":"On Accelerating Edge AI: Optimizing Resource-Constrained Environments","date":"2025-01-25","arxiv_id":"2501.15014","repositories_listed":0,"syntology":null},{"url":null,"slug":"rotatekv-accurate-and-robust-2-bit-kv-cache","title":"RotateKV: Accurate and Robust 2-Bit KV Cache Quantization for LLMs via Outlier-Aware Adaptive Rotations","date":"2025-01-25","arxiv_id":"2501.16383","repositories_listed":0,"syntology":null},{"url":null,"slug":"channel-aware-constellation-design-for","title":"Channel-Aware Constellation Design for Digital OTA Computation","date":"2025-01-24","arxiv_id":"2501.14675","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-workflow-for-machine-learning","title":"End-to-end workflow for machine learning-based qubit readout with QICK and hls4ml","date":"2025-01-24","arxiv_id":"2501.14663","repositories_listed":0,"syntology":null},{"url":null,"slug":"hwpq-hessian-free-weight-pruning-quantization","title":"SwiftPrune: Hessian-Free Weight Pruning for Large Language Models","date":"2025-01-24","arxiv_id":"2501.16376","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-hardening-dnns-against-noisy-computations","title":"On Hardening DNNs against Noisy Computations","date":"2025-01-24","arxiv_id":"2501.14531","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-based-perceptual-neural-video","title":"Diffusion-based Perceptual Neural Video Compression with Temporal Diffusion Information Reuse","date":"2025-01-23","arxiv_id":"2501.13528","repositories_listed":0,"syntology":null},{"url":null,"slug":"dq-data2vec-decoupling-quantization-for","title":"DQ-Data2vec: Decoupling Quantization for Multilingual Speech Recognition","date":"2025-01-23","arxiv_id":"2501.13497","repositories_listed":0,"syntology":null},{"url":null,"slug":"mambaquant-quantizing-the-mamba-family-with","title":"MambaQuant: Quantizing the Mamba Family with Variance Aligned Rotation Methods","date":"2025-01-23","arxiv_id":"2501.13484","repositories_listed":0,"syntology":null},{"url":null,"slug":"qmamba-post-training-quantization-for-vision","title":"QMamba: Post-Training Quantization for Vision State Space Models","date":"2025-01-23","arxiv_id":"2501.13624","repositories_listed":0,"syntology":null},{"url":null,"slug":"qrazor-reliable-and-effortless-4-bit-llm","title":"Qrazor: Reliable and effortless 4-bit llm quantization by significant data razoring","date":"2025-01-23","arxiv_id":"2501.13331","repositories_listed":0,"syntology":null},{"url":null,"slug":"heppo-hardware-efficient-proximal-policy","title":"HEPPO: Hardware-Efficient Proximal Policy Optimization -- A Universal Pipelined Architecture for Generalized Advantage Estimation","date":"2025-01-22","arxiv_id":"2501.12703","repositories_listed":0,"syntology":null},{"url":null,"slug":"irrational-complex-rotations-empower-low-bit","title":"Irrational Complex Rotations Empower Low-bit Optimizers","date":"2025-01-22","arxiv_id":"2501.12896","repositories_listed":0,"syntology":null},{"url":null,"slug":"sketch-and-patch-efficient-3d-gaussian","title":"Sketch and Patch: Efficient 3D Gaussian Representation for Man-Made Scenes","date":"2025-01-22","arxiv_id":"2501.13045","repositories_listed":0,"syntology":null},{"url":null,"slug":"rl-rc-dot-a-block-level-rl-agent-for-task","title":"RL-RC-DoT: A Block-level RL agent for Task-Aware Video Compression","date":"2025-01-21","arxiv_id":"2501.12216","repositories_listed":0,"syntology":null},{"url":null,"slug":"splitquant-layer-splitting-for-low-bit-neural","title":"SplitQuant: Layer Splitting for Low-Bit Neural Network Quantization","date":"2025-01-21","arxiv_id":"2501.12428","repositories_listed":0,"syntology":null},{"url":null,"slug":"uav-assisted-real-time-disaster-detection","title":"UAV-Assisted Real-Time Disaster Detection Using Optimized Transformer Model","date":"2025-01-21","arxiv_id":"2501.12087","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-efficient-federated-learning-by","title":"Communication-Efficient Federated Learning by Quantized Variance Reduction for Heterogeneous Wireless Edge Networks","date":"2025-01-20","arxiv_id":"2501.11267","repositories_listed":0,"syntology":null},{"url":null,"slug":"ditto-accelerating-diffusion-model-via","title":"Ditto: Accelerating Diffusion Model via Temporal Value Similarity","date":"2025-01-20","arxiv_id":"2501.11211","repositories_listed":0,"syntology":null},{"url":null,"slug":"personalized-federated-learning-for-cellular","title":"Personalized Federated Learning for Cellular VR: Online Learning and Dynamic Caching","date":"2025-01-20","arxiv_id":"2501.11745","repositories_listed":0,"syntology":null},{"url":null,"slug":"practical-modulo-sampling-mitigating-high","title":"Practical Modulo Sampling: Mitigating High-Frequency Components","date":"2025-01-20","arxiv_id":"2501.11330","repositories_listed":0,"syntology":null},{"url":null,"slug":"best-a-novel-source-selection-metric-for","title":"BeST -- A Novel Source Selection Metric for Transfer Learning","date":"2025-01-19","arxiv_id":"2501.10933","repositories_listed":0,"syntology":null},{"url":null,"slug":"dc-pcn-point-cloud-completion-network-with","title":"DC-PCN: Point Cloud Completion Network with Dual-Codebook Guided Quantization","date":"2025-01-19","arxiv_id":"2501.10966","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-novel-hybrid-precoder-with-low-resolution","title":"A Novel Hybrid Precoder With Low-Resolution Phase Shifters and Fronthaul Capacity Limitation","date":"2025-01-18","arxiv_id":"2501.10878","repositories_listed":0,"syntology":null},{"url":null,"slug":"lut-dla-lookup-table-as-efficient-extreme-low","title":"LUT-DLA: Lookup Table as Efficient Extreme Low-Bit Deep Learning Accelerator","date":"2025-01-18","arxiv_id":"2501.10658","repositories_listed":0,"syntology":null},{"url":null,"slug":"atleus-accelerating-transformers-on-the-edge","title":"Atleus: Accelerating Transformers on the Edge Enabled by 3D Heterogeneous Manycore Architectures","date":"2025-01-16","arxiv_id":"2501.09588","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-devil-is-in-the-details-simple-remedies","title":"The Devil is in the Details: Simple Remedies for Image-to-LiDAR Representation Learning","date":"2025-01-16","arxiv_id":"2501.09485","repositories_listed":0,"syntology":null},{"url":null,"slug":"real-time-indexing-for-large-scale","title":"Real-time Indexing for Large-scale Recommendation by Streaming Vector Quantization Retriever","date":"2025-01-15","arxiv_id":"2501.08695","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-post-training-quantization","title":"Rethinking Post-Training Quantization: Introducing a Statistical Pre-Calibration Approach","date":"2025-01-15","arxiv_id":"2501.09107","repositories_listed":0,"syntology":null},{"url":null,"slug":"large-language-models-for-text-classification","title":"Large Language Models For Text Classification: Case Study And Comprehensive Review","date":"2025-01-14","arxiv_id":"2501.08457","repositories_listed":0,"syntology":null}],"record_sha256":"8b0a1f1b6d353343097de9b52049997ca2595b24401d68315825d87b987f625c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}