{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/quantization/papers/25","list_of":"/task/quantization","task":"Quantization","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":25,"pages_in_order":50,"rows_per_page":100,"rows":[2401,2500],"of":4925,"counts":{"archive_papers_tagged":4925,"with_a_code_link":1596,"where_syntology_ran_a_sample":515,"not_listed_spam_title":0,"listed":4925,"listed_where_code_ran":515,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":452,"every_run_a_failure_of_syntologys_instrument":63,"listed_with_a_run_with_no_instrument_failure":452,"listed_every_run_a_failure_of_syntologys_instrument":63,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/quantization","prev":"/task/quantization/papers/24","next":"/task/quantization/papers/26","papers":[{"url":null,"slug":"quality-scalable-quantization-methodology-for","title":"Quality Scalable Quantization Methodology for Deep Learning on Edge","date":"2024-07-15","arxiv_id":"2407.11260","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-bag-of-tricks-for-scaling-cpu-based-deep","title":"A Bag of Tricks for Scaling CPU-based Deep FFMs to more than 300m Predictions per Second","date":"2024-07-14","arxiv_id":"2407.10115","repositories_listed":0,"syntology":null},{"url":"/paper/leanquant-accurate-large-language-model","slug":"leanquant-accurate-large-language-model","title":"LeanQuant: Accurate Large Language Model Quantization with Loss-Error-Aware Grid","date":"2024-07-14","arxiv_id":"2407.10032","repositories_listed":0,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/leanquant-accurate-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2407.10032","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10032"}},"official":null}},{"url":null,"slug":"one-bit-mimo-detection-from-global-maximum","title":"One-Bit MIMO Detection: From Global Maximum-Likelihood Detector to Amplitude Retrieval Approach","date":"2024-07-13","arxiv_id":"2407.11079","repositories_listed":0,"syntology":null},{"url":null,"slug":"accuracy-is-not-all-you-need","title":"Accuracy is Not All You Need","date":"2024-07-12","arxiv_id":"2407.09141","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimization-of-dnn-based-speaker","title":"Optimization of DNN-based speaker verification model through efficient quantization technique","date":"2024-07-12","arxiv_id":"2407.08991","repositories_listed":0,"syntology":null},{"url":null,"slug":"admm-based-semi-structured-pattern-pruning","title":"ADMM Based Semi-Structured Pattern Pruning Framework For Transformer","date":"2024-07-11","arxiv_id":"2407.08334","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoregressive-speech-synthesis-without","title":"Autoregressive Speech Synthesis without Vector Quantization","date":"2024-07-11","arxiv_id":"2407.08551","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-deep-reinforcement-learning-based","title":"Distributed Deep Reinforcement Learning Based Gradient Quantization for Federated Learning Enabled Vehicle Edge Computing","date":"2024-07-11","arxiv_id":"2407.08462","repositories_listed":0,"syntology":null},{"url":null,"slug":"erq-error-reduction-for-post-training","title":"ERQ: Error Reduction for Post-Training Quantization of Vision Transformers","date":"2024-07-09","arxiv_id":"2407.06794","repositories_listed":0,"syntology":null},{"url":null,"slug":"ternary-spike-based-neuromorphic-signal","title":"Ternary Spike-based Neuromorphic Signal Processing System","date":"2024-07-07","arxiv_id":"2407.05310","repositories_listed":0,"syntology":null},{"url":null,"slug":"balance-of-number-of-embedding-and-their","title":"Balance of Number of Embedding and their Dimensions in Vector Quantization","date":"2024-07-06","arxiv_id":"2407.04939","repositories_listed":0,"syntology":null},{"url":null,"slug":"integer-only-quantized-transformers-for","title":"Integer-only Quantized Transformers for Embedded FPGA-based Time-series Forecasting in AIoT","date":"2024-07-06","arxiv_id":"2407.11041","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantizing-yolov7-a-comprehensive-study","title":"Quantizing YOLOv7: A Comprehensive Study","date":"2024-07-06","arxiv_id":"2407.04943","repositories_listed":0,"syntology":null},{"url":null,"slug":"zobnn-zero-overhead-dependable-design-of","title":"ZOBNN: Zero-Overhead Dependable Design of Binary Neural Networks with Deliberately Quantized Parameters","date":"2024-07-06","arxiv_id":"2407.04964","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-receiver-design-for-massive-mimo-ofdm","title":"Hybrid Receiver Design for Massive MIMO-OFDM with Low-Resolution ADCs and Oversampling","date":"2024-07-05","arxiv_id":"2407.04408","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-quantization-and-pruning-on","title":"The Impact of Quantization and Pruning on Deep Reinforcement Learning Models","date":"2024-07-05","arxiv_id":"2407.04803","repositories_listed":0,"syntology":null},{"url":null,"slug":"hera-high-efficiency-matrix-compression-via","title":"QET: Enhancing Quantized LLM Parameters and KV cache Compression through Element Substitution and Residual Clustering","date":"2024-07-04","arxiv_id":"2407.03637","repositories_listed":0,"syntology":null},{"url":null,"slug":"joint-beamforming-design-and-bit-allocation","title":"Joint Beamforming Design and Bit Allocation in Massive MIMO with Resolution-Adaptive ADCs","date":"2024-07-04","arxiv_id":"2407.03796","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-latency-machine-learning-fpga-accelerator","title":"Low-latency machine learning FPGA accelerator for multi-qubit-state discrimination","date":"2024-07-04","arxiv_id":"2407.03852","repositories_listed":0,"syntology":null},{"url":null,"slug":"timestep-aware-correction-for-quantized","title":"Timestep-Aware Correction for Quantized Diffusion Models","date":"2024-07-04","arxiv_id":"2407.03917","repositories_listed":0,"syntology":null},{"url":null,"slug":"adfq-vit-activation-distribution-friendly","title":"ADFQ-ViT: Activation-Distribution-Friendly Post-Training Quantization for Vision Transformers","date":"2024-07-03","arxiv_id":"2407.02763","repositories_listed":0,"syntology":null},{"url":null,"slug":"codec-asr-training-performant-automatic","title":"Codec-ASR: Training Performant Automatic Speech Recognition Systems with Discrete Speech Representations","date":"2024-07-03","arxiv_id":"2407.03495","repositories_listed":0,"syntology":null},{"url":null,"slug":"edge-ai-enabled-chicken-health-detection","title":"Edge AI-Enabled Chicken Health Detection Based on Enhanced FCOS-Lite and Knowledge Distillation","date":"2024-07-03","arxiv_id":"2407.09562","repositories_listed":0,"syntology":null},{"url":null,"slug":"fisher-aware-quantization-for-detr-detectors","title":"Fisher-aware Quantization for DETR Detectors with Critical-category Objectives","date":"2024-07-03","arxiv_id":"2407.03442","repositories_listed":0,"syntology":null},{"url":null,"slug":"gptqt-quantize-large-language-models-twice-to","title":"GPTQT: Quantize Large Language Models Twice to Push the Efficiency","date":"2024-07-03","arxiv_id":"2407.02891","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-does-quantization-affect-multilingual","title":"How Does Quantization Affect Multilingual LLMs?","date":"2024-07-03","arxiv_id":"2407.03211","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-conversational-abilities-of","title":"Improving Conversational Abilities of Quantized Large Language Models via Direct Preference Alignment","date":"2024-07-03","arxiv_id":"2407.03051","repositories_listed":0,"syntology":null},{"url":null,"slug":"ospc-artificial-vlm-features-for-hateful-meme","title":"OSPC: Artificial VLM Features for Hateful Meme Detection","date":"2024-07-03","arxiv_id":"2407.12836","repositories_listed":0,"syntology":null},{"url":null,"slug":"sfc-achieve-accurate-fast-convolution-under","title":"SFC: Achieve Accurate Fast Convolution under Low-precision Arithmetic","date":"2024-07-03","arxiv_id":"2407.02913","repositories_listed":0,"syntology":null},{"url":null,"slug":"unified-anomaly-detection-methods-on-edge","title":"Unified Anomaly Detection methods on Edge Device using Knowledge Distillation and Quantization","date":"2024-07-03","arxiv_id":"2407.02968","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-throughput-and-compression-ratios","title":"Beyond Throughput and Compression Ratios: Towards High End-to-end Utility of Gradient Compression","date":"2024-07-01","arxiv_id":"2407.01378","repositories_listed":0,"syntology":null},{"url":null,"slug":"exploring-fpga-designs-for-mx-and-beyond","title":"Exploring FPGA designs for MX and beyond","date":"2024-07-01","arxiv_id":"2407.01475","repositories_listed":0,"syntology":null},{"url":null,"slug":"linear-and-nonlinear-mmse-estimation-in-one","title":"Linear and Nonlinear MMSE Estimation in One-Bit Quantized Systems under a Gaussian Mixture Prior","date":"2024-07-01","arxiv_id":"2407.01305","repositories_listed":0,"syntology":null},{"url":null,"slug":"pqcache-product-quantization-based-kvcache","title":"PQCache: Product Quantization-based KVCache for Long Context LLM Inference","date":"2024-07-01","arxiv_id":"2407.12820","repositories_listed":0,"syntology":null},{"url":null,"slug":"hasnas-a-hardware-aware-spiking-neural","title":"NeuroNAS: Enhancing Efficiency of Neuromorphic In-Memory Computing for Intelligent Mobile Agents through Hardware-Aware Spiking Neural Architecture Search","date":"2024-06-30","arxiv_id":"2407.00641","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-a-diffusion-based-generalist-for-dense","title":"Toward a Diffusion-Based Generalist for Dense Vision Tasks","date":"2024-06-29","arxiv_id":"2407.00503","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-fusion-model-for-brain-tumor","title":"Deep Fusion Model for Brain Tumor Classification Using Fine-Grained Gradient Preservation","date":"2024-06-28","arxiv_id":"2406.19690","repositories_listed":0,"syntology":null},{"url":null,"slug":"rateless-stochastic-coding-for-delay","title":"Rateless Stochastic Coding for Delay-Constrained Semantic Communication","date":"2024-06-28","arxiv_id":"2406.19804","repositories_listed":0,"syntology":null},{"url":null,"slug":"fronthaul-quantization-aware-mu-mimo","title":"Fronthaul Quantization-Aware MU-MIMO Precoding for Sum Rate Maximization","date":"2024-06-27","arxiv_id":"2406.19183","repositories_listed":0,"syntology":null},{"url":null,"slug":"mcnc-manifold-constrained-network-compression","title":"MCNC: Manifold Constrained Network Compression","date":"2024-06-27","arxiv_id":"2406.19301","repositories_listed":0,"syntology":null},{"url":null,"slug":"outliertune-efficient-channel-wise","title":"OutlierTune: Efficient Channel-Wise Quantization for Large Language Models","date":"2024-06-27","arxiv_id":"2406.18832","repositories_listed":0,"syntology":null},{"url":null,"slug":"reliable-edge-machine-learning-hardware-for","title":"Reliable edge machine learning hardware for scientific applications","date":"2024-06-27","arxiv_id":"2406.19522","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-quantization-based-technique-for-privacy","title":"A Quantization-based Technique for Privacy Preserving Distributed Learning","date":"2024-06-26","arxiv_id":"2406.19418","repositories_listed":0,"syntology":null},{"url":null,"slug":"differential-error-feedback-for-communication","title":"Differential error feedback for communication-efficient decentralized learning","date":"2024-06-26","arxiv_id":"2406.18418","repositories_listed":0,"syntology":null},{"url":null,"slug":"fedaq-communication-efficient-federated-edge","title":"FedAQ: Communication-Efficient Federated Edge Learning via Joint Uplink and Downlink Adaptive Quantization","date":"2024-06-26","arxiv_id":"2406.18156","repositories_listed":0,"syntology":null},{"url":null,"slug":"cdquant-accurate-post-training-weight","title":"CDQuant: Greedy Coordinate Descent for Accurate LLM Quantization","date":"2024-06-25","arxiv_id":"2406.17542","repositories_listed":0,"syntology":null},{"url":null,"slug":"approximate-dct-and-quantization-techniques","title":"Approximate DCT and Quantization Techniques for Energy-Constrained Image Sensors","date":"2024-06-24","arxiv_id":"2406.16358","repositories_listed":0,"syntology":null},{"url":null,"slug":"bitnet-b1-58-reloaded-state-of-the-art","title":"BitNet b1.58 Reloaded: State-of-the-art Performance Also on Smaller Networks","date":"2024-06-24","arxiv_id":"2407.09527","repositories_listed":0,"syntology":null},{"url":null,"slug":"compensate-quantization-errors-make-weights","title":"Compensate Quantization Errors: Make Weights Hierarchical to Compensate Each Other","date":"2024-06-24","arxiv_id":"2406.16299","repositories_listed":0,"syntology":null},{"url":null,"slug":"leveraging-knowledge-distillation-for-1","title":"Leveraging Knowledge Distillation for Lightweight Skin Cancer Classification: Balancing Accuracy and Computational Efficiency","date":"2024-06-24","arxiv_id":"2406.17051","repositories_listed":0,"syntology":null},{"url":null,"slug":"reducing-the-memory-footprint-of-3d-gaussian","title":"Reducing the Memory Footprint of 3D Gaussian Splatting","date":"2024-06-24","arxiv_id":"2406.17074","repositories_listed":0,"syntology":null},{"url":null,"slug":"received-power-maximization-using-nonuniform","title":"Received Power Maximization Using Nonuniform Discrete Phase Shifts for RISs With a Limited Phase Range","date":"2024-06-23","arxiv_id":"2406.16210","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-real-time-neural-volumetric-rendering","title":"Towards Real-Time Neural Volumetric Rendering on Mobile Devices: A Measurement Study","date":"2024-06-23","arxiv_id":"2406.16068","repositories_listed":0,"syntology":null},{"url":null,"slug":"hlq-fast-and-efficient-backpropagation-via","title":"HLQ: Fast and Efficient Backpropagation via Hadamard Low-rank Quantization","date":"2024-06-21","arxiv_id":"2406.15102","repositories_listed":0,"syntology":null},{"url":null,"slug":"predicting-probabilities-of-error-to-combine","title":"Predicting Probabilities of Error to Combine Quantization and Early Exiting: QuEE","date":"2024-06-20","arxiv_id":"2406.14404","repositories_listed":0,"syntology":null},{"url":"/paper/attention-aware-post-training-quantization","slug":"attention-aware-post-training-quantization","title":"Attention-aware Post-training Quantization without Backpropagation","date":"2024-06-19","arxiv_id":"2406.13474","repositories_listed":0,"syntology":{"n":11,"n_ran":10,"n_constructed":1,"n_ran_checked":1,"n_instrument":9,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":11,"phrase":"10 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 9 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/attention-aware-post-training-quantization#ran","syntology_url":"https://syntology.ai/paper/2406.13474","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13474"}},"official":null}},{"url":null,"slug":"high-fidelity-facial-albedo-estimation-via","title":"High-Fidelity Facial Albedo Estimation via Texture Quantization","date":"2024-06-19","arxiv_id":"2406.13149","repositories_listed":0,"syntology":null},{"url":null,"slug":"q-snns-quantized-spiking-neural-networks","title":"Q-SNNs: Quantized Spiking Neural Networks","date":"2024-06-19","arxiv_id":"2406.13672","repositories_listed":0,"syntology":null},{"url":null,"slug":"sdq-sparse-decomposed-quantization-for-llm","title":"SDQ: Sparse Decomposed Quantization for LLM Inference","date":"2024-06-19","arxiv_id":"2406.13868","repositories_listed":0,"syntology":null},{"url":null,"slug":"bayesian-lora-lora-based-parameter-efficient","title":"Bayesian-LoRA: LoRA based Parameter Efficient Fine-Tuning using Optimal Quantization levels and Rank Values trough Differentiable Bayesian Gates","date":"2024-06-18","arxiv_id":"2406.13046","repositories_listed":0,"syntology":null},{"url":null,"slug":"mse-minimization-in-ris-aided-mu-mimo-with","title":"MSE Minimization in RIS-Aided MU-MIMO with Discrete Phase Shifts and Fronthaul Quantization","date":"2024-06-18","arxiv_id":"2406.12388","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-learning-based-channel-estimation-for-7","title":"Deep-Learning-Based Channel Estimation for Distributed MIMO with 1-bit Radio-Over-Fiber Fronthaul","date":"2024-06-17","arxiv_id":"2406.11325","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-analysis-on-quantizing-diffusion","title":"An Analysis on Quantizing Diffusion Transformers","date":"2024-06-16","arxiv_id":"2406.11100","repositories_listed":0,"syntology":null},{"url":null,"slug":"promoting-data-and-model-privacy-in-federated","title":"Promoting Data and Model Privacy in Federated Learning through Quantized LoRA","date":"2024-06-16","arxiv_id":"2406.10976","repositories_listed":0,"syntology":null},{"url":null,"slug":"tender-accelerating-large-language-models-via","title":"Tender: Accelerating Large Language Models via Tensor Decomposition and Runtime Requantization","date":"2024-06-16","arxiv_id":"2406.12930","repositories_listed":0,"syntology":null},{"url":null,"slug":"how-should-we-extract-discrete-audio-tokens","title":"How Should We Extract Discrete Audio Tokens from Self-Supervised Models?","date":"2024-06-15","arxiv_id":"2406.10735","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-faults-in-activation-sparse-quantized","title":"Memory Faults in Activation-sparse Quantized Deep Neural Networks: Analysis and Mitigation using Sharpness-aware Training","date":"2024-06-15","arxiv_id":"2406.10528","repositories_listed":0,"syntology":null},{"url":null,"slug":"geb-1-3b-open-lightweight-large-language","title":"GEB-1.3B: Open Lightweight Large Language Model","date":"2024-06-14","arxiv_id":"2406.09900","repositories_listed":0,"syntology":null},{"url":null,"slug":"one-pass-multiple-conformer-and-foundation","title":"One-pass Multiple Conformer and Foundation Speech Systems Compression and Quantization Using An All-in-one Neural Model","date":"2024-06-14","arxiv_id":"2406.10160","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-byte-level-representation-for-end","title":"Optimizing Byte-level Representation for End-to-end ASR","date":"2024-06-14","arxiv_id":"2406.09676","repositories_listed":0,"syntology":null},{"url":null,"slug":"precipitation-nowcasting-using-physics","title":"Precipitation Nowcasting Using Physics Informed Discriminator Generative Models","date":"2024-06-14","arxiv_id":"2406.10108","repositories_listed":0,"syntology":null},{"url":null,"slug":"human-level-molecular-optimization-driven-by","title":"Human-level molecular optimization driven by mol-gene evolution","date":"2024-06-13","arxiv_id":"2406.12910","repositories_listed":0,"syntology":null},{"url":null,"slug":"me-switch-a-memory-efficient-expert-switching","title":"ME-Switch: A Memory-Efficient Expert Switching Framework for Large Language Models","date":"2024-06-13","arxiv_id":"2406.09041","repositories_listed":0,"syntology":null},{"url":null,"slug":"mgrq-post-training-quantization-for-vision","title":"MGRQ: Post-Training Quantization For Vision Transformer With Mixed Granularity Reconstruction","date":"2024-06-13","arxiv_id":"2406.09229","repositories_listed":0,"syntology":null},{"url":null,"slug":"toneunit-a-speech-discretization-approach-for","title":"ToneUnit: A Speech Discretization Approach for Tonal Language Speech Synthesis","date":"2024-06-13","arxiv_id":"2406.08989","repositories_listed":0,"syntology":null},{"url":null,"slug":"asymptotic-unbiased-sample-sampling-to-speed","title":"Asymptotic Unbiased Sample Sampling to Speed Up Sharpness-Aware Minimization","date":"2024-06-12","arxiv_id":"2406.08001","repositories_listed":0,"syntology":null},{"url":null,"slug":"compressive-beam-alignment-for-indoor","title":"Compressive Beam Alignment for Indoor Millimeter-Wave Systems","date":"2024-06-12","arxiv_id":"2406.07965","repositories_listed":0,"syntology":null},{"url":null,"slug":"mobileaibench-benchmarking-llms-and-lmms-for","title":"MobileAIBench: Benchmarking LLMs and LMMs for On-Device Use Cases","date":"2024-06-12","arxiv_id":"2406.10290","repositories_listed":0,"syntology":null},{"url":null,"slug":"vall-e-r-robust-and-efficient-zero-shot-text","title":"VALL-E R: Robust and Efficient Zero-Shot Text-to-Speech Synthesis via Monotonic Alignment","date":"2024-06-12","arxiv_id":"2406.07855","repositories_listed":0,"syntology":null},{"url":null,"slug":"foldtoken2-learning-compact-invariant-and","title":"FoldToken2: Learning compact, invariant and generative protein structure language","date":"2024-06-11","arxiv_id":"2407.00050","repositories_listed":0,"syntology":null},{"url":null,"slug":"t2s-gpt-dynamic-vector-quantization-for","title":"T2S-GPT: Dynamic Vector Quantization for Autoregressive Sign Language Production from Text","date":"2024-06-11","arxiv_id":"2406.07119","repositories_listed":0,"syntology":null},{"url":null,"slug":"ternaryllm-ternarized-large-language-model","title":"TernaryLLM: Ternarized Large Language Model","date":"2024-06-11","arxiv_id":"2406.07177","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-neural-compression-with-inference","title":"Efficient Neural Compression with Inference-time Decoding","date":"2024-06-10","arxiv_id":"2406.06237","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-representation-matters-human-like","title":"Latent Representation Matters: Human-like Sketches in One-shot Drawing Tasks","date":"2024-06-10","arxiv_id":"2406.06079","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-impact-of-quantization-on-retrieval","title":"The Impact of Quantization on Retrieval-Augmented Generation: An Analysis of Small LLMs","date":"2024-06-10","arxiv_id":"2406.10251","repositories_listed":0,"syntology":null},{"url":null,"slug":"topological-analysis-for-detecting-anomalies","title":"Topological Analysis for Detecting Anomalies (TADA) in Time Series","date":"2024-06-10","arxiv_id":"2406.06168","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-lightweight-speaker-verification-via","title":"Towards Lightweight Speaker Verification via Adaptive Neural Network Quantization","date":"2024-06-08","arxiv_id":"2406.05359","repositories_listed":0,"syntology":null},{"url":null,"slug":"activation-map-based-vector-quantization-for","title":"Activation Map-based Vector Quantization for 360-degree Image Semantic Communication","date":"2024-06-07","arxiv_id":"2406.04740","repositories_listed":0,"syntology":null},{"url":null,"slug":"spectral-codecs-spectrogram-based-audio","title":"Spectral Codecs: Improving Non-Autoregressive Speech Synthesis with Spectrogram-Based Audio Codecs","date":"2024-06-07","arxiv_id":"2406.05298","repositories_listed":0,"syntology":null},{"url":null,"slug":"proofread-fixes-all-errors-with-one-tap","title":"Proofread: Fixes All Errors with One Tap","date":"2024-06-06","arxiv_id":"2406.04523","repositories_listed":0,"syntology":null},{"url":null,"slug":"usm-rnn-t-model-weights-binarization","title":"USM RNN-T model weights binarization","date":"2024-06-05","arxiv_id":"2406.02887","repositories_listed":0,"syntology":null},{"url":null,"slug":"vqunet-vector-quantization-u-net-for","title":"VQUNet: Vector Quantization U-Net for Defending Adversarial Atacks by Regularizing Unwanted Noise","date":"2024-06-05","arxiv_id":"2406.03117","repositories_listed":0,"syntology":null},{"url":null,"slug":"zeroth-order-fine-tuning-of-llms-with-extreme","title":"Zeroth-Order Fine-Tuning of LLMs with Extreme Sparsity","date":"2024-06-05","arxiv_id":"2406.02913","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixed-precision-over-the-air-federated","title":"Mixed-Precision Federated Learning via Multi-Precision Over-The-Air Aggregation","date":"2024-06-04","arxiv_id":"2406.03402","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-efficient-deep-spiking-neuron-networks","title":"Toward Efficient Deep Spiking Neuron Networks:A Survey On Compression","date":"2024-06-03","arxiv_id":"2407.08744","repositories_listed":0,"syntology":null},{"url":null,"slug":"log-scale-quantization-in-distributed-first","title":"Log-Scale Quantization in Distributed First-Order Methods: Gradient-based Learning from Distributed Data","date":"2024-06-02","arxiv_id":"2406.00621","repositories_listed":0,"syntology":null},{"url":null,"slug":"effective-interplay-between-sparsity-and","title":"Effective Interplay between Sparsity and Quantization: From Theory to Practice","date":"2024-05-31","arxiv_id":"2405.20935","repositories_listed":0,"syntology":null},{"url":null,"slug":"lcq-low-rank-codebook-based-quantization-for","title":"LCQ: Low-Rank Codebook based Quantization for Large Language Models","date":"2024-05-31","arxiv_id":"2405.20973","repositories_listed":0,"syntology":null},{"url":null,"slug":"locking-machine-learning-models-into-hardware","title":"Locking Machine Learning Models into Hardware","date":"2024-05-31","arxiv_id":"2405.20990","repositories_listed":0,"syntology":null}],"record_sha256":"d8aaea31eaac4d20a3e05c2e6155f15d99721789e5f54bf54e02c0ef78095335","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}