{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/quantization/papers/18","list_of":"/task/quantization","task":"Quantization","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":18,"pages_in_order":50,"rows_per_page":100,"rows":[1701,1800],"of":4925,"counts":{"archive_papers_tagged":4925,"with_a_code_link":1596,"where_syntology_ran_a_sample":515,"not_listed_spam_title":0,"listed":4925,"listed_where_code_ran":515,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":452,"every_run_a_failure_of_syntologys_instrument":63,"listed_with_a_run_with_no_instrument_failure":452,"listed_every_run_a_failure_of_syntologys_instrument":63,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/quantization","prev":"/task/quantization/papers/17","next":"/task/quantization/papers/19","papers":[{"url":null,"slug":"unihm-universal-human-motion-generation-with","title":"UniHM: Universal Human Motion Generation with Object Interactions in Indoor Scenes","date":"2025-05-19","arxiv_id":"2505.12774","repositories_listed":0,"syntology":null},{"url":null,"slug":"calm-co-evolution-of-algorithms-and-language","title":"CALM: Co-evolution of Algorithms and Language Model for Automatic Heuristic Design","date":"2025-05-18","arxiv_id":"2505.12285","repositories_listed":0,"syntology":null},{"url":null,"slug":"hyperbolic-residual-quantization-discrete","title":"Hyperbolic Residual Quantization: Discrete Representations for Data with Latent Hierarchies","date":"2025-05-18","arxiv_id":"2505.12404","repositories_listed":0,"syntology":null},{"url":null,"slug":"kvmix-gradient-based-layer-importance-aware","title":"KVmix: Gradient-Based Layer Importance-Aware Mixed-Precision Quantization for KV Cache","date":"2025-05-18","arxiv_id":"2506.08018","repositories_listed":0,"syntology":null},{"url":null,"slug":"fedhq-hybrid-runtime-quantization-for","title":"FedHQ: Hybrid Runtime Quantization for Federated Learning","date":"2025-05-17","arxiv_id":"2505.11982","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11170","title":"Gaussian Weight Sampling for Scalable, Efficient and Stable Pseudo-Quantization Training","date":"2025-05-16","arxiv_id":"2505.11170","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-11334","title":"MARRS: Masked Autoregressive Unit-based Reaction Synthesis","date":"2025-05-16","arxiv_id":"2505.11334","repositories_listed":0,"syntology":null},{"url":"/paper/2505-11497","slug":"2505-11497","title":"QVGen: Pushing the Limit of Quantized Video Generative Models","date":"2025-05-16","arxiv_id":"2505.11497","repositories_listed":0,"syntology":{"n":7,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/2505-11497#ran","syntology_url":"https://syntology.ai/paper/2505.11497","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.11497"}},"official":null}},{"url":null,"slug":"benchmarking-cfar-and-cnn-based-peak","title":"Benchmarking CFAR and CNN-based Peak Detection Algorithms in ISAC under Hardware Impairments","date":"2025-05-16","arxiv_id":"2505.10969","repositories_listed":0,"syntology":null},{"url":null,"slug":"formal-uncertainty-propagation-for-stochastic","title":"Formal Uncertainty Propagation for Stochastic Dynamical Systems with Additive Noise","date":"2025-05-16","arxiv_id":"2505.11219","repositories_listed":0,"syntology":null},{"url":null,"slug":"qronos-correcting-the-past-by-shaping-the","title":"Qronos: Correcting the Past by Shaping the Future... in Post-Training Quantization","date":"2025-05-16","arxiv_id":"2505.11695","repositories_listed":0,"syntology":null},{"url":null,"slug":"2505-10689","title":"A probabilistic framework for dynamic quantization","date":"2025-05-15","arxiv_id":"2505.10689","repositories_listed":0,"syntology":null},{"url":null,"slug":"vq-logits-compressing-the-output-bottleneck","title":"VQ-Logits: Compressing the Output Bottleneck of Large Language Models via Vector Quantized Logits","date":"2025-05-15","arxiv_id":"2505.10202","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-quantization-a-comprehensive-survey","title":"Zero-shot Quantization: A Comprehensive Survey","date":"2025-05-14","arxiv_id":"2505.09188","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-layer-hierarchical-federated-learning","title":"Multi-Layer Hierarchical Federated Learning with Quantization","date":"2025-05-13","arxiv_id":"2505.08145","repositories_listed":0,"syntology":null},{"url":null,"slug":"resource-efficient-language-models","title":"Resource-Efficient Language Models: Quantization for Fast and Accessible Inference","date":"2025-05-13","arxiv_id":"2505.08620","repositories_listed":0,"syntology":null},{"url":null,"slug":"an-extra-rmsnorm-is-all-you-need-for-fine","title":"An Extra RMSNorm is All You Need for Fine Tuning to 1.58 Bits","date":"2025-05-12","arxiv_id":"2505.08823","repositories_listed":0,"syntology":null},{"url":null,"slug":"bang-for-the-buck-vector-search-on-cloud-cpus","title":"Bang for the Buck: Vector Search on Cloud CPUs","date":"2025-05-12","arxiv_id":"2505.07621","repositories_listed":0,"syntology":null},{"url":null,"slug":"cognitive-non-coherent-jamming-techniques-for","title":"Cognitive Non-Coherent Jamming Techniques for Frequency Selective Attacks","date":"2025-05-12","arxiv_id":"2505.07429","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-ann-snn-conversion-with-error","title":"Efficient ANN-SNN Conversion with Error Compensation Learning","date":"2025-05-12","arxiv_id":"2506.01968","repositories_listed":0,"syntology":null},{"url":null,"slug":"private-lora-fine-tuning-of-open-source-llms","title":"Private LoRA Fine-tuning of Open-Source LLMs with Homomorphic Encryption","date":"2025-05-12","arxiv_id":"2505.07329","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantx-a-framework-for-hardware-aware","title":"QuantX: A Framework for Hardware-Aware Quantization of Generative AI Workloads","date":"2025-05-12","arxiv_id":"2505.07531","repositories_listed":0,"syntology":null},{"url":null,"slug":"semantic-retention-and-extreme-compression-in","title":"Semantic Retention and Extreme Compression in LLMs: Can We Have Both?","date":"2025-05-12","arxiv_id":"2505.07289","repositories_listed":0,"syntology":null},{"url":null,"slug":"improving-block-wise-llm-quantization-by-4","title":"Improving Block-Wise LLM Quantization by 4-bit Block-Wise Optimal Float (BOF4): Analysis and Variations","date":"2025-05-10","arxiv_id":"2505.06653","repositories_listed":0,"syntology":null},{"url":null,"slug":"challenging-gpu-dominance-when-cpus","title":"Challenging GPU Dominance: When CPUs Outperform for On-Device LLM Inference","date":"2025-05-09","arxiv_id":"2505.06461","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightnobel-improving-sequence-length","title":"LightNobel: Improving Sequence Length Limitation in Protein Structure Prediction Model via Adaptive Activation Quantization","date":"2025-05-09","arxiv_id":"2505.05893","repositories_listed":0,"syntology":null},{"url":null,"slug":"turbo-icl-in-context-learning-based-turbo","title":"Turbo-ICL: In-Context Learning-Based Turbo Equalization","date":"2025-05-09","arxiv_id":"2505.06175","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-loss-landscape-generalizable","title":"Learning from Loss Landscape: Generalizable Mixed-Precision Quantization via Adaptive Sharpness-Aware Gradient Aligning","date":"2025-05-08","arxiv_id":"2505.04877","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-bit-model-quantization-for-deep-neural","title":"Low-bit Model Quantization for Deep Neural Networks: A Survey","date":"2025-05-08","arxiv_id":"2505.05530","repositories_listed":0,"syntology":null},{"url":null,"slug":"mix-qsam-mixed-precision-quantization-of-the","title":"Mix-QSAM: Mixed-Precision Quantization of the Segment Anything Model","date":"2025-05-08","arxiv_id":"2505.04861","repositories_listed":0,"syntology":null},{"url":null,"slug":"reactdance-progressive-granular","title":"ReactDance: Progressive-Granular Representation for Long-Term Coherent Reactive Dance Generation","date":"2025-05-08","arxiv_id":"2505.05589","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-gaussian-splatting-data-compression-with","title":"3D Gaussian Splatting Data Compression with Mixture of Priors","date":"2025-05-06","arxiv_id":"2505.03310","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-clinical-decision-support-system","title":"Lightweight Clinical Decision Support System using QLoRA-Fine-Tuned LLMs and Retrieval-Augmented Generation","date":"2025-05-06","arxiv_id":"2505.03406","repositories_listed":0,"syntology":null},{"url":null,"slug":"prom-prioritize-reduction-of-multiplications","title":"PROM: Prioritize Reduction of Multiplications Over Lower Bit-Widths for Efficient CNNs","date":"2025-05-06","arxiv_id":"2505.03254","repositories_listed":0,"syntology":null},{"url":null,"slug":"bielik-11b-v2-technical-report","title":"Bielik 11B v2 Technical Report","date":"2025-05-05","arxiv_id":"2505.02410","repositories_listed":0,"syntology":null},{"url":null,"slug":"end-to-end-fully-binarized-network-design","title":"End-to-end fully-binarized network design: from Generic Learned Thermometer to Block Pruning","date":"2025-05-05","arxiv_id":"2505.13462","repositories_listed":0,"syntology":null},{"url":null,"slug":"entrollm-entropy-encoded-weight-compression","title":"EntroLLM: Entropy Encoded Weight Compression for Efficient Large Language Model Inference on Edge Devices","date":"2025-05-05","arxiv_id":"2505.02380","repositories_listed":0,"syntology":null},{"url":null,"slug":"neurosim-v1-5-improved-software-backbone-for","title":"NeuroSim V1.5: Improved Software Backbone for Benchmarking Compute-in-Memory Accelerators with Device and Circuit-level Non-idealities","date":"2025-05-05","arxiv_id":"2505.02314","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-llms-for-resource-constrained","title":"Optimizing LLMs for Resource-Constrained Environments: A Survey of Model Compression Techniques","date":"2025-05-05","arxiv_id":"2505.02309","repositories_listed":0,"syntology":null},{"url":"/paper/radio-rate-distortion-optimization-for-large","slug":"radio-rate-distortion-optimization-for-large","title":"Radio: Rate-Distortion Optimization for Large Language Model Compression","date":"2025-05-05","arxiv_id":"2505.03031","repositories_listed":0,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/radio-rate-distortion-optimization-for-large#ran","syntology_url":"https://syntology.ai/paper/2505.03031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.03031"}},"official":null}},{"url":null,"slug":"rapid-yet-accurate-tile-circuit-and-device","title":"Rapid yet accurate Tile-circuit and device modeling for Analog In-Memory Computing","date":"2025-05-05","arxiv_id":"2506.00004","repositories_listed":0,"syntology":null},{"url":null,"slug":"robsurv-vector-quantization-based-multi-modal","title":"RobSurv: Vector Quantization-Based Multi-Modal Learning for Robust Cancer Survival Prediction","date":"2025-05-05","arxiv_id":"2505.02529","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantizing-diffusion-models-from-a-sampling","title":"Quantizing Diffusion Models from a Sampling-Aware Perspective","date":"2025-05-04","arxiv_id":"2505.02242","repositories_listed":0,"syntology":null},{"url":null,"slug":"pascal-precise-and-efficient-ann-snn","title":"PASCAL: Precise and Efficient ANN- SNN Conversion using Spike Accumulation and Adaptive Layerwise Activation","date":"2025-05-03","arxiv_id":"2505.01730","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-fine-tuning-of-quantized-models-via","title":"Efficient Fine-Tuning of Quantized Models via Adaptive Rank and Bitwidth","date":"2025-05-02","arxiv_id":"2505.03802","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-vision-based-vehicle-speed","title":"Efficient Vision-based Vehicle Speed Estimation","date":"2025-05-02","arxiv_id":"2505.01203","repositories_listed":0,"syntology":null},{"url":null,"slug":"grouped-sequency-arranged-rotation-optimizing","title":"Grouped Sequency-arranged Rotation: Optimizing Rotation Transformation for Quantization for Free","date":"2025-05-02","arxiv_id":"2505.03810","repositories_listed":0,"syntology":null},{"url":null,"slug":"lmdepth-lightweight-mamba-based-monocular","title":"LMDepth: Lightweight Mamba-based Monocular Depth Estimation for Real-World Deployment","date":"2025-05-02","arxiv_id":"2505.00980","repositories_listed":0,"syntology":null},{"url":null,"slug":"aggregating-empirical-evidence-from-data","title":"Aggregating empirical evidence from data strategy studies: a case on model quantization","date":"2025-05-01","arxiv_id":"2505.00816","repositories_listed":0,"syntology":null},{"url":null,"slug":"generative-qoe-modeling-a-lightweight","title":"Generative QoE Modeling: A Lightweight Approach for Telecom Networks","date":"2025-04-30","arxiv_id":"2504.21353","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimization-of-embeddings-storage-for-rag","title":"Optimization of embeddings storage for RAG systems using quantization and dimensionality reduction techniques","date":"2025-04-30","arxiv_id":"2505.00105","repositories_listed":0,"syntology":null},{"url":null,"slug":"precision-where-it-matters-a-novel-spike","title":"Precision Where It Matters: A Novel Spike Aware Mixed-Precision Quantization Strategy for LLaMA-based Language Models","date":"2025-04-30","arxiv_id":"2504.21553","repositories_listed":0,"syntology":null},{"url":null,"slug":"apg-mos-auditory-perception-guided-mos","title":"APG-MOS: Auditory Perception Guided-MOS Predictor for Synthetic Speech","date":"2025-04-29","arxiv_id":"2504.20447","repositories_listed":0,"syntology":null},{"url":null,"slug":"clustering-based-evolutionary-federated","title":"Clustering-Based Evolutionary Federated Multiobjective Optimization and Learning","date":"2025-04-29","arxiv_id":"2504.20346","repositories_listed":0,"syntology":null},{"url":null,"slug":"fineq-software-hardware-co-design-for-low-bit","title":"FineQ: Software-Hardware Co-Design for Low-Bit Fine-Grained Mixed-Precision Quantization of LLMs","date":"2025-04-28","arxiv_id":"2504.19746","repositories_listed":0,"syntology":null},{"url":null,"slug":"turboquant-online-vector-quantization-with","title":"TurboQuant: Online Vector Quantization with Near-optimal Distortion Rate","date":"2025-04-28","arxiv_id":"2504.19874","repositories_listed":0,"syntology":null},{"url":null,"slug":"pushing-the-boundary-on-natural-language","title":"Pushing the boundary on Natural Language Inference","date":"2025-04-25","arxiv_id":"2504.18376","repositories_listed":0,"syntology":null},{"url":null,"slug":"fast-autoregressive-models-for-continuous","title":"Fast Autoregressive Models for Continuous Latent Generation","date":"2025-04-24","arxiv_id":"2504.18391","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-device-qwen2-5-efficient-llm-inference","title":"On-Device Qwen2.5: Efficient LLM Inference with Model Compression and Hardware Acceleration","date":"2025-04-24","arxiv_id":"2504.17376","repositories_listed":0,"syntology":null},{"url":null,"slug":"precision-neural-network-quantization-via","title":"Precision Neural Network Quantization via Learnable Adaptive Modules","date":"2025-04-24","arxiv_id":"2504.17263","repositories_listed":0,"syntology":null},{"url":null,"slug":"distributed-optimization-with-efficient","title":"Distributed Optimization with Efficient Communication, Event-Triggered Solution Enhancement, and Operation Stopping","date":"2025-04-23","arxiv_id":"2504.16477","repositories_listed":0,"syntology":null},{"url":null,"slug":"hexcute-a-tile-based-programming-language","title":"Hexcute: A Tile-based Programming Language with Automatic Layout and Task-Mapping Synthesis","date":"2025-04-22","arxiv_id":"2504.16214","repositories_listed":0,"syntology":null},{"url":null,"slug":"tellme-an-energy-efficient-ternary-llm","title":"TeLLMe: An Energy-Efficient Ternary LLM Accelerator for Prefilling and Decoding on Edge FPGAs","date":"2025-04-22","arxiv_id":"2504.16266","repositories_listed":0,"syntology":null},{"url":null,"slug":"compute-optimal-llms-provably-generalize","title":"Compute-Optimal LLMs Provably Generalize Better With Scale","date":"2025-04-21","arxiv_id":"2504.15208","repositories_listed":0,"syntology":null},{"url":null,"slug":"stablequant-layer-adaptive-post-training","title":"StableQuant: Layer Adaptive Post-Training Quantization for Speech Foundation Models","date":"2025-04-21","arxiv_id":"2504.14915","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-implicit-neural-compression-of","title":"Efficient Implicit Neural Compression of Point Clouds via Learnable Activation in Latent Space","date":"2025-04-20","arxiv_id":"2504.14471","repositories_listed":0,"syntology":null},{"url":null,"slug":"fgmp-fine-grained-mixed-precision-weight-and","title":"FGMP: Fine-Grained Mixed-Precision Weight and Activation Quantization for Hardware-Accelerated LLM Inference","date":"2025-04-19","arxiv_id":"2504.14152","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-road-environment-segmentation","title":"Lightweight Road Environment Segmentation using Vector Quantization","date":"2025-04-19","arxiv_id":"2504.14113","repositories_listed":0,"syntology":null},{"url":null,"slug":"from-large-to-super-tiny-end-to-end","title":"From Large to Super-Tiny: End-to-End Optimization for Cost-Efficient LLMs","date":"2025-04-18","arxiv_id":"2504.13471","repositories_listed":0,"syntology":null},{"url":null,"slug":"gradual-binary-search-and-dimension-expansion","title":"Gradual Binary Search and Dimension Expansion : A general method for activation quantization in LLMs","date":"2025-04-18","arxiv_id":"2504.13989","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-binary-and-ternary-quantization-can","title":"The Binary and Ternary Quantization Can Improve Feature Discrimination","date":"2025-04-18","arxiv_id":"2504.13792","repositories_listed":0,"syntology":null},{"url":null,"slug":"d-2-moe-dual-routing-and-dynamic-scheduling","title":"D$^{2}$MoE: Dual Routing and Dynamic Scheduling for Efficient On-Device MoE-based LLM Serving","date":"2025-04-17","arxiv_id":"2504.15299","repositories_listed":0,"syntology":null},{"url":null,"slug":"fedx-adaptive-model-decomposition-and","title":"FedX: Adaptive Model Decomposition and Quantization for IoT Federated Learning","date":"2025-04-17","arxiv_id":"2504.12849","repositories_listed":0,"syntology":null},{"url":null,"slug":"esc-mvq-end-to-end-semantic-communication","title":"ESC-MVQ: End-to-End Semantic Communication With Multi-Codebook Vector Quantization","date":"2025-04-16","arxiv_id":"2504.11709","repositories_listed":0,"syntology":null},{"url":null,"slug":"resume-abstractif-a-partir-d-une","title":"Résumé abstractif à partir d'une transcription audio","date":"2025-04-16","arxiv_id":"2504.11803","repositories_listed":0,"syntology":null},{"url":null,"slug":"csplade-learned-sparse-retrieval-with-causal","title":"CSPLADE: Learned Sparse Retrieval with Causal Language Models","date":"2025-04-15","arxiv_id":"2504.10816","repositories_listed":0,"syntology":null},{"url":null,"slug":"goat-tts-llm-based-text-to-speech-generation","title":"GOAT-TTS: Expressive and Realistic Speech Generation via A Dual-Branch LLM","date":"2025-04-15","arxiv_id":"2504.12339","repositories_listed":0,"syntology":null},{"url":null,"slug":"neural-network-emulation-of-the-classical","title":"Neural Network Emulation of the Classical Limit in Quantum Systems via Learned Observable Mappings","date":"2025-04-15","arxiv_id":"2504.10781","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantization-error-propagation-revisiting","title":"Quantization Error Propagation: Revisiting Layer-Wise Post-Training Quantization","date":"2025-04-13","arxiv_id":"2504.09629","repositories_listed":0,"syntology":null},{"url":null,"slug":"simultaneous-input-and-state-estimation-under","title":"Simultaneous Input and State Estimation under Output Quantization: A Gaussian Mixture approach","date":"2025-04-13","arxiv_id":"2504.09711","repositories_listed":0,"syntology":null},{"url":null,"slug":"asymptotic-stabilization-under-homomorphic","title":"Asymptotic stabilization under homomorphic encryption: A re-encryption free method","date":"2025-04-12","arxiv_id":"2504.09248","repositories_listed":0,"syntology":null},{"url":null,"slug":"deploying-large-ai-models-on-resource-limited","title":"Deploying Large AI Models on Resource-Limited Devices with Split Federated Learning","date":"2025-04-12","arxiv_id":"2504.09114","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixdit-accelerating-image-diffusion","title":"MixDiT: Accelerating Image Diffusion Transformer Inference with Mixed-Precision MX Quantization","date":"2025-04-11","arxiv_id":"2504.08398","repositories_listed":0,"syntology":null},{"url":null,"slug":"motiondreamer-one-to-many-motion-synthesis","title":"MotionDreamer: One-to-Many Motion Synthesis with Localized Generative Masked Transformer","date":"2025-04-11","arxiv_id":"2504.08959","repositories_listed":0,"syntology":null},{"url":null,"slug":"muon-accelerated-attention-distillation-for","title":"Muon-Accelerated Attention Distillation for Real-Time Edge Synthesis via Optimized Latent Diffusion","date":"2025-04-11","arxiv_id":"2504.08451","repositories_listed":0,"syntology":null},{"url":null,"slug":"specee-accelerating-large-language-model","title":"SpecEE: Accelerating Large Language Model Inference with Speculative Early Exiting","date":"2025-04-11","arxiv_id":"2504.08850","repositories_listed":0,"syntology":null},{"url":null,"slug":"apsq-additive-partial-sum-quantization-with","title":"APSQ: Additive Partial Sum Quantization with Algorithm-Hardware Co-Design","date":"2025-04-10","arxiv_id":"2505.03748","repositories_listed":0,"syntology":null},{"url":null,"slug":"pogo-a-scalable-proof-of-useful-work-via","title":"PoGO: A Scalable Proof of Useful Work via Quantized Gradient Descent and Merkle Proofs","date":"2025-04-10","arxiv_id":"2504.07540","repositories_listed":0,"syntology":null},{"url":null,"slug":"bbqrec-behavior-bind-quantization-for-multi","title":"BBQRec: Behavior-Bind Quantization for Multi-Modal Sequential Recommendation","date":"2025-04-09","arxiv_id":"2504.06636","repositories_listed":0,"syntology":null},{"url":null,"slug":"chime-a-compressive-framework-for-holistic","title":"CHIME: A Compressive Framework for Holistic Interest Modeling","date":"2025-04-09","arxiv_id":"2504.06780","repositories_listed":0,"syntology":null},{"url":null,"slug":"accllm-accelerating-long-context-llm","title":"AccLLM: Accelerating Long-Context LLM Inference Via Algorithm-Hardware Co-Design","date":"2025-04-07","arxiv_id":"2505.03745","repositories_listed":0,"syntology":null},{"url":null,"slug":"achieving-binary-weight-and-activation-for","title":"Achieving binary weight and activation for LLMs using Post-Training Quantization","date":"2025-04-07","arxiv_id":"2504.05352","repositories_listed":0,"syntology":null},{"url":null,"slug":"balancing-robustness-and-efficiency-in","title":"Balancing Robustness and Efficiency in Embedded DNNs Through Activation Function Selection","date":"2025-04-07","arxiv_id":"2504.05119","repositories_listed":0,"syntology":null},{"url":null,"slug":"bridging-the-gap-between-continuous-and","title":"Bridging the Gap between Continuous and Informative Discrete Representations by Random Product Quantization","date":"2025-04-07","arxiv_id":"2504.04721","repositories_listed":0,"syntology":null},{"url":null,"slug":"two-is-better-than-one-efficient-ensemble","title":"Two is Better than One: Efficient Ensemble Defense for Robust and Compact Models","date":"2025-04-07","arxiv_id":"2504.04747","repositories_listed":0,"syntology":null},{"url":null,"slug":"skin-color-measurement-from-dermatoscopic","title":"Skin Color Measurement from Dermatoscopic Images: An Evaluation on a Synthetic Dataset","date":"2025-04-06","arxiv_id":"2504.04494","repositories_listed":0,"syntology":null},{"url":null,"slug":"autoregressive-high-order-finite-difference","title":"Autoregressive High-Order Finite Difference Modulo Imaging: High-Dynamic Range for Computer Vision Applications","date":"2025-04-05","arxiv_id":"2504.04228","repositories_listed":0,"syntology":null},{"url":null,"slug":"efficient-fpga-accelerated-convolutional","title":"Efficient FPGA-accelerated Convolutional Neural Networks for Cloud Detection on CubeSats","date":"2025-04-04","arxiv_id":"2504.03891","repositories_listed":0,"syntology":null},{"url":null,"slug":"shape-my-moves-text-driven-shape-aware","title":"Shape My Moves: Text-Driven Shape-Aware Synthesis of Human Motions","date":"2025-04-04","arxiv_id":"2504.03639","repositories_listed":0,"syntology":null},{"url":null,"slug":"sustainable-llm-inference-for-edge-ai","title":"Sustainable LLM Inference for Edge AI: Evaluating Quantized LLMs for Energy Efficiency, Output Accuracy, and Inference Latency","date":"2025-04-04","arxiv_id":"2504.03360","repositories_listed":0,"syntology":null}],"record_sha256":"1a4a88709f7306253fd311bb51c3d726754cbbfc8e2f6e10834cb548e48f01f4","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}