{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/quantization/papers/22","list_of":"/task/quantization","task":"Quantization","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":22,"pages_in_order":50,"rows_per_page":100,"rows":[2101,2200],"of":4925,"counts":{"archive_papers_tagged":4925,"with_a_code_link":1596,"where_syntology_ran_a_sample":515,"not_listed_spam_title":0,"listed":4925,"listed_where_code_ran":515,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":452,"every_run_a_failure_of_syntologys_instrument":63,"listed_with_a_run_with_no_instrument_failure":452,"listed_every_run_a_failure_of_syntologys_instrument":63,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/quantization","prev":"/task/quantization/papers/21","next":"/task/quantization/papers/23","papers":[{"url":null,"slug":"sensor-selection-and-distributed-quantization","title":"Sensor Selection and Distributed Quantization for Energy Efficiency in Massive MTC","date":"2024-12-07","arxiv_id":"2412.05626","repositories_listed":0,"syntology":null},{"url":null,"slug":"trimming-down-large-spiking-vision","title":"Trimming Down Large Spiking Vision Transformers via Heterogeneous Quantization Search","date":"2024-12-07","arxiv_id":"2412.05505","repositories_listed":0,"syntology":null},{"url":null,"slug":"ulmrec-user-centric-large-language-model-for","title":"ULMRec: User-centric Large Language Model for Sequential Recommendation","date":"2024-12-07","arxiv_id":"2412.05543","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantized-and-interpretable-learning-scheme","title":"Quantized and Interpretable Learning Scheme for Deep Neural Networks in Classification Task","date":"2024-12-05","arxiv_id":"2412.03915","repositories_listed":0,"syntology":null},{"url":null,"slug":"skim-any-bit-quantization-pushing-the-limits","title":"SKIM: Any-bit Quantization Pushing The Limits of Post-Training Quantization","date":"2024-12-05","arxiv_id":"2412.04180","repositories_listed":0,"syntology":null},{"url":null,"slug":"designing-dnns-for-a-trade-off-between","title":"Designing DNNs for a trade-off between robustness and processing performance in embedded devices","date":"2024-12-04","arxiv_id":"2412.03682","repositories_listed":0,"syntology":null},{"url":null,"slug":"flashattention-on-a-napkin-a-diagrammatic","title":"FlashAttention on a Napkin: A Diagrammatic Approach to Deep Learning IO-Awareness","date":"2024-12-04","arxiv_id":"2412.03317","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixed-precision-quantization-make-the-best","title":"Mixed-Precision Quantization: Make the Best Use of Bits Where They Matter Most","date":"2024-12-04","arxiv_id":"2412.03101","repositories_listed":0,"syntology":null},{"url":null,"slug":"prompting-large-language-models-for-clinical","title":"Prompting Large Language Models for Clinical Temporal Relation Extraction","date":"2024-12-04","arxiv_id":"2412.04512","repositories_listed":0,"syntology":null},{"url":null,"slug":"unifying-kv-cache-compression-for-large","title":"Unifying KV Cache Compression for Large Language Models with LeanKV","date":"2024-12-04","arxiv_id":"2412.03131","repositories_listed":0,"syntology":null},{"url":null,"slug":"3d-representation-in-512-byte-variational","title":"3D representation in 512-Byte:Variational tokenizer is the key for autoregressive 3D generation","date":"2024-12-03","arxiv_id":"2412.02202","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-classic-quantum-hybrid-network-framework","title":"Lean classical-quantum hybrid neural network model for image classification","date":"2024-12-03","arxiv_id":"2412.02059","repositories_listed":0,"syntology":null},{"url":null,"slug":"cegi-measuring-the-trade-off-between","title":"CEGI: Measuring the trade-off between efficiency and carbon emissions for SLMs and VLMs","date":"2024-12-03","arxiv_id":"2412.02602","repositories_listed":0,"syntology":null},{"url":null,"slug":"cptquant-a-novel-mixed-precision-post","title":"CPTQuant -- A Novel Mixed Precision Post-Training Quantization Techniques for Large Language Models","date":"2024-12-03","arxiv_id":"2412.03599","repositories_listed":0,"syntology":null},{"url":null,"slug":"robust-precoding-for-multi-user-visible-light","title":"Robust Precoding for Multi-User Visible Light Communications with Quantized Channel Information","date":"2024-12-03","arxiv_id":"2412.02219","repositories_listed":0,"syntology":null},{"url":null,"slug":"memory-efficient-training-for-deep-speaker","title":"Memory-Efficient Training for Deep Speaker Embedding Learning in Speaker Verification","date":"2024-12-02","arxiv_id":"2412.01195","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-domain-specific-image-retrieval-a","title":"Optimizing Domain-Specific Image Retrieval: A Benchmark of FAISS and Annoy with Fine-Tuned Features","date":"2024-12-02","arxiv_id":"2412.01555","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantization-aware-imitation-learning-for","title":"Quantization-Aware Imitation-Learning for Resource-Efficient Robotic Control","date":"2024-12-02","arxiv_id":"2412.01034","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-wave-is-worth-100-words-investigating-cross","title":"A Wave is Worth 100 Words: Investigating Cross-Domain Transferability in Time Series","date":"2024-12-01","arxiv_id":"2412.00772","repositories_listed":0,"syntology":null},{"url":null,"slug":"lambda-covering-the-multimodal-critical","title":"LAMBDA: Covering the Multimodal Critical Scenarios for Automated Driving Systems by Search Space Quantization","date":"2024-11-30","arxiv_id":"2412.00517","repositories_listed":0,"syntology":null},{"url":null,"slug":"cogact-a-foundational-vision-language-action","title":"CogACT: A Foundational Vision-Language-Action Model for Synergizing Cognition and Action in Robotic Manipulation","date":"2024-11-29","arxiv_id":"2411.19650","repositories_listed":0,"syntology":null},{"url":"/paper/discord-discrete-tokens-to-continuous-motion","slug":"discord-discrete-tokens-to-continuous-motion","title":"DisCoRD: Discrete Tokens to Continuous Motion via Rectified Flow Decoding","date":"2024-11-29","arxiv_id":"2411.19527","repositories_listed":0,"syntology":null},{"url":null,"slug":"privacy-preserving-orthogonal-aggregation-for","title":"Privacy-Preserving Orthogonal Aggregation for Guaranteeing Gender Fairness in Federated Recommendation","date":"2024-11-29","arxiv_id":"2411.19678","repositories_listed":0,"syntology":null},{"url":null,"slug":"quantized-delta-weight-is-safety-keeper","title":"Quantized Delta Weight Is Safety Keeper","date":"2024-11-29","arxiv_id":"2411.19530","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-effectiveness-of-discrete","title":"On the effectiveness of discrete representations in sparse mixture of experts","date":"2024-11-28","arxiv_id":"2411.19402","repositories_listed":0,"syntology":null},{"url":null,"slug":"orthus-autoregressive-interleaved-image-text","title":"Orthus: Autoregressive Interleaved Image-Text Generation with Modality-Specific Heads","date":"2024-11-28","arxiv_id":"2412.00127","repositories_listed":0,"syntology":null},{"url":null,"slug":"fames-fast-approximate-multiplier","title":"FAMES: Fast Approximate Multiplier Substitution for Mixed-Precision Quantized DNNs--Down to 2 Bits!","date":"2024-11-27","arxiv_id":"2411.18055","repositories_listed":0,"syntology":null},{"url":null,"slug":"coap-memory-efficient-training-with","title":"COAP: Memory-Efficient Training with Correlation-Aware Gradient Projection","date":"2024-11-26","arxiv_id":"2412.00071","repositories_listed":0,"syntology":null},{"url":null,"slug":"low-bit-quantization-favors-undertrained-llms","title":"Low-Bit Quantization Favors Undertrained LLMs: Scaling Laws for Quantized LLMs with 100T Training Tokens","date":"2024-11-26","arxiv_id":"2411.17691","repositories_listed":0,"syntology":null},{"url":null,"slug":"rapid-deployment-of-domain-specific","title":"Rapid Deployment of Domain-specific Hyperspectral Image Processors with Application to Autonomous Driving","date":"2024-11-26","arxiv_id":"2411.17543","repositories_listed":0,"syntology":null},{"url":null,"slug":"softmap-software-hardware-co-design-for","title":"SoftmAP: Software-Hardware Co-design for Integer-Only Softmax on Associative Processors","date":"2024-11-26","arxiv_id":"2411.17847","repositories_listed":0,"syntology":null},{"url":null,"slug":"beyond-task-vectors-selective-task-arithmetic","title":"Beyond Task Vectors: Selective Task Arithmetic Based on Importance Metrics","date":"2024-11-25","arxiv_id":"2411.16139","repositories_listed":0,"syntology":null},{"url":null,"slug":"curvature-in-the-looking-glass-optimal","title":"Curvature in the Looking-Glass: Optimal Methods to Exploit Curvature of Expectation in the Loss Landscape","date":"2024-11-25","arxiv_id":"2411.16914","repositories_listed":0,"syntology":null},{"url":null,"slug":"downlink-mimo-channel-estimation-from-bits","title":"Downlink MIMO Channel Estimation from Bits: Recoverability and Algorithm","date":"2024-11-25","arxiv_id":"2411.16043","repositories_listed":0,"syntology":null},{"url":null,"slug":"factorized-visual-tokenization-and-generation","title":"Factorized Visual Tokenization and Generation","date":"2024-11-25","arxiv_id":"2411.16681","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-optimal-lattice-vector-quantizers","title":"Learning Optimal Lattice Vector Quantizers for End-to-end Neural Image Compression","date":"2024-11-25","arxiv_id":"2411.16119","repositories_listed":0,"syntology":null},{"url":null,"slug":"lion-cub-minimizing-communication-overhead-in","title":"Lion Cub: Minimizing Communication Overhead in Distributed Lion","date":"2024-11-25","arxiv_id":"2411.16462","repositories_listed":0,"syntology":null},{"url":null,"slug":"mixpe-quantization-and-hardware-co-design-for","title":"MixPE: Quantization and Hardware Co-design for Efficient LLM Inference","date":"2024-11-25","arxiv_id":"2411.16158","repositories_listed":0,"syntology":null},{"url":null,"slug":"representation-collapsing-problems-in-vector","title":"Representation Collapsing Problems in Vector Quantization","date":"2024-11-25","arxiv_id":"2411.16550","repositories_listed":0,"syntology":null},{"url":null,"slug":"rethinking-diffusion-for-text-driven-human","title":"Rethinking Diffusion for Text-Driven Human Motion Generation","date":"2024-11-25","arxiv_id":"2411.16575","repositories_listed":0,"syntology":null},{"url":null,"slug":"skqvc-one-shot-voice-conversion-by-k-means","title":"SKQVC: One-Shot Voice Conversion by K-Means Quantization with Self-Supervised Speech Representations","date":"2024-11-25","arxiv_id":"2411.16147","repositories_listed":0,"syntology":null},{"url":null,"slug":"freepruner-a-training-free-approach-for-large","title":"freePruner: A Training-free Approach for Large Multimodal Model Acceleration","date":"2024-11-23","arxiv_id":"2411.15446","repositories_listed":0,"syntology":null},{"url":null,"slug":"flare-fp-less-ptq-and-low-enob-adc-based-ams","title":"FLARE: FP-Less PTQ and Low-ENOB ADC Based AMS-PiM for Error-Resilient, Fast, and Efficient Transformer Acceleration","date":"2024-11-22","arxiv_id":"2411.14733","repositories_listed":0,"syntology":null},{"url":null,"slug":"automixq-self-adjusting-quantization-for-high","title":"AutoMixQ: Self-Adjusting Quantization for High Performance Memory-Efficient Fine-Tuning","date":"2024-11-21","arxiv_id":"2411.13814","repositories_listed":0,"syntology":null},{"url":"/paper/taq-dit-time-aware-quantization-for-diffusion","slug":"taq-dit-time-aware-quantization-for-diffusion","title":"TaQ-DiT: Time-aware Quantization for Diffusion Transformers","date":"2024-11-21","arxiv_id":"2411.14172","repositories_listed":0,"syntology":{"n":7,"n_ran":5,"n_constructed":2,"n_ran_checked":2,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/taq-dit-time-aware-quantization-for-diffusion#ran","syntology_url":"https://syntology.ai/paper/2411.14172","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14172"}},"official":null}},{"url":null,"slug":"disco-intelligent-omni-surfaces-360-degree","title":"Disco Intelligent Omni-Surfaces: 360-degree Fully-Passive Jamming Attacks","date":"2024-11-20","arxiv_id":"2411.12985","repositories_listed":0,"syntology":null},{"url":null,"slug":"rtsr-a-real-time-super-resolution-model-for","title":"RTSR: A Real-Time Super-Resolution Model for AV1 Compressed Content","date":"2024-11-20","arxiv_id":"2411.13362","repositories_listed":0,"syntology":null},{"url":null,"slug":"diffusion-product-quantization","title":"Diffusion Product Quantization","date":"2024-11-19","arxiv_id":"2411.12306","repositories_listed":0,"syntology":null},{"url":null,"slug":"high-throughput-blind-co-channel-interference","title":"High-Throughput Blind Co-Channel Interference Cancellation for Edge Devices Using Depthwise Separable Convolutions, Quantization, and Pruning","date":"2024-11-19","arxiv_id":"2411.12541","repositories_listed":0,"syntology":null},{"url":null,"slug":"efqat-an-efficient-framework-for-quantization","title":"EfQAT: An Efficient Framework for Quantization-Aware Training","date":"2024-11-17","arxiv_id":"2411.11038","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-accurate-and-efficient-sub-8-bit","title":"Towards Accurate and Efficient Sub-8-Bit Integer Training","date":"2024-11-17","arxiv_id":"2411.10948","repositories_listed":0,"syntology":null},{"url":null,"slug":"bluelm-v-3b-algorithm-and-system-co-design","title":"BlueLM-V-3B: Algorithm and System Co-Design for Multimodal Large Language Models on Mobile Devices","date":"2024-11-16","arxiv_id":"2411.10640","repositories_listed":0,"syntology":null},{"url":null,"slug":"amxfp4-taming-activation-outliers-with","title":"AMXFP4: Taming Activation Outliers with Asymmetric Microscaling Floating-Point for 4-bit LLM Inference","date":"2024-11-15","arxiv_id":"2411.09909","repositories_listed":0,"syntology":null},{"url":null,"slug":"systolic-arrays-and-structured-pruning-co","title":"Systolic Arrays and Structured Pruning Co-design for Efficient Transformers in Edge Systems","date":"2024-11-15","arxiv_id":"2411.10285","repositories_listed":0,"syntology":null},{"url":null,"slug":"communication-compression-for-tensor-parallel","title":"Communication Compression for Tensor Parallel LLM Inference","date":"2024-11-14","arxiv_id":"2411.09510","repositories_listed":0,"syntology":null},{"url":null,"slug":"aser-activation-smoothing-and-error","title":"ASER: Activation Smoothing and Error Reconstruction for Large Language Model Quantization","date":"2024-11-12","arxiv_id":"2411.07762","repositories_listed":0,"syntology":null},{"url":null,"slug":"navigation-with-qphil-quantizing-planner-for","title":"Navigation with QPHIL: Quantizing Planner for Hierarchical Implicit Q-Learning","date":"2024-11-12","arxiv_id":"2411.07760","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-low-bit-communication-for-tensor","title":"Towards Low-bit Communication for Tensor Parallel LLM Inference","date":"2024-11-12","arxiv_id":"2411.07942","repositories_listed":0,"syntology":null},{"url":null,"slug":"harmlevelbench-evaluating-harm-level","title":"HarmLevelBench: Evaluating Harm-Level Compliance and the Impact of Quantization on Model Alignment","date":"2024-11-11","arxiv_id":"2411.06835","repositories_listed":0,"syntology":null},{"url":null,"slug":"sketched-adaptive-federated-deep-learning-a","title":"Sketched Adaptive Federated Deep Learning: A Sharp Convergence Analysis","date":"2024-11-11","arxiv_id":"2411.06770","repositories_listed":0,"syntology":null},{"url":null,"slug":"federated-llms-fine-tuned-with-adaptive","title":"HAFLQ: Heterogeneous Adaptive Federated LoRA Fine-tuned LLM with Quantization","date":"2024-11-10","arxiv_id":"2411.06581","repositories_listed":0,"syntology":null},{"url":null,"slug":"intelligent-fault-diagnosis-of-type-and","title":"Intelligent Fault Diagnosis of Type and Severity in Low-Frequency, Low Bit-Depth Signals","date":"2024-11-09","arxiv_id":"2411.06299","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-large-language-models-through","title":"Optimizing Large Language Models through Quantization: A Comparative Analysis of PTQ and QAT Techniques","date":"2024-11-09","arxiv_id":"2411.06084","repositories_listed":0,"syntology":null},{"url":"/paper/aligned-vector-quantization-for-edge-cloud","slug":"aligned-vector-quantization-for-edge-cloud","title":"Aligned Vector Quantization for Edge-Cloud Collabrative Vision-Language Models","date":"2024-11-08","arxiv_id":"2411.05961","repositories_listed":0,"syntology":null},{"url":null,"slug":"quancrypt-fl-quantized-homomorphic-encryption","title":"QuanCrypt-FL: Quantized Homomorphic Encryption with Pruning for Secure Federated Learning","date":"2024-11-08","arxiv_id":"2411.05260","repositories_listed":0,"syntology":null},{"url":null,"slug":"qwen2-5-32b-leveraging-self-consistent-tool","title":"Qwen2.5-32B: Leveraging Self-Consistent Tool-Integrated Reasoning for Bengali Mathematical Olympiad Problem Solving","date":"2024-11-08","arxiv_id":"2411.05934","repositories_listed":0,"syntology":null},{"url":null,"slug":"rate-aware-compression-for-nerf-based","title":"Rate-aware Compression for NeRF-based Volumetric Video","date":"2024-11-08","arxiv_id":"2411.05322","repositories_listed":0,"syntology":null},{"url":null,"slug":"when-are-1-58-bits-enough-a-bottom-up","title":"When are 1.58 bits enough? A Bottom-up Exploration of BitNet Quantization","date":"2024-11-08","arxiv_id":"2411.05882","repositories_listed":0,"syntology":null},{"url":null,"slug":"compressive-spectrum-sensing-with-1-bit-adcs","title":"Compressive Spectrum Sensing with 1-bit ADCs","date":"2024-11-07","arxiv_id":"2411.04611","repositories_listed":0,"syntology":null},{"url":null,"slug":"green-my-llm-studying-the-key-factors","title":"Green My LLM: Studying the key factors affecting the energy consumption of code assistants","date":"2024-11-07","arxiv_id":"2411.11892","repositories_listed":0,"syntology":null},{"url":null,"slug":"saliency-assisted-quantization-for-neural","title":"Saliency Assisted Quantization for Neural Networks","date":"2024-11-07","arxiv_id":"2411.05858","repositories_listed":0,"syntology":null},{"url":null,"slug":"interactions-across-blocks-in-post-training","title":"Interactions Across Blocks in Post-Training Quantization of Large Language Models","date":"2024-11-06","arxiv_id":"2411.03934","repositories_listed":0,"syntology":null},{"url":null,"slug":"multi-bit-distributed-detection-of-sparse","title":"Multi-bit Distributed Detection of Sparse Stochastic Signals over Error-Prone Reporting Channels","date":"2024-11-06","arxiv_id":"2411.03612","repositories_listed":0,"syntology":null},{"url":null,"slug":"hybrid-beamforming-for-integrated-sensing-and","title":"Hybrid Beamforming for Integrated Sensing and Communications With Low Resolution DACs","date":"2024-11-05","arxiv_id":"2411.02827","repositories_listed":0,"syntology":null},{"url":null,"slug":"sum-rate-maximization-in-the-constant","title":"Sum Rate Maximization in the Constant Envelope MIMO Downlink with the RZF Precoder","date":"2024-11-05","arxiv_id":"2411.03084","repositories_listed":0,"syntology":null},{"url":null,"slug":"give-me-bf16-or-give-me-death-accuracy","title":"\"Give Me BF16 or Give Me Death\"? Accuracy-Performance Trade-Offs in LLM Quantization","date":"2024-11-04","arxiv_id":"2411.02355","repositories_listed":0,"syntology":null},{"url":null,"slug":"transferable-sequential-recommendation-via","title":"Transferable Sequential Recommendation via Vector Quantized Meta Learning","date":"2024-11-04","arxiv_id":"2411.01785","repositories_listed":0,"syntology":null},{"url":null,"slug":"bf-imna-a-bit-fluid-in-memory-neural","title":"BF-IMNA: A Bit Fluid In-Memory Neural Architecture for Neural Network Acceleration","date":"2024-11-03","arxiv_id":"2411.01417","repositories_listed":0,"syntology":null},{"url":null,"slug":"fundamental-trade-offs-in-quantized-hybrid","title":"Fundamental Trade-offs in Quantized Hybrid Radar Fusion: A CRB-Rate Perspective","date":"2024-11-01","arxiv_id":"2411.00496","repositories_listed":0,"syntology":null},{"url":null,"slug":"optimizing-contextual-speech-recognition","title":"Optimizing Contextual Speech Recognition Using Vector Quantization for Efficient Retrieval","date":"2024-11-01","arxiv_id":"2411.00664","repositories_listed":0,"syntology":null},{"url":null,"slug":"alise-accelerating-large-language-model","title":"ALISE: Accelerating Large Language Model Serving with Speculative Scheduling","date":"2024-10-31","arxiv_id":"2410.23537","repositories_listed":0,"syntology":null},{"url":null,"slug":"arq-a-mixed-precision-quantization-framework","title":"ARQ: A Mixed-Precision Quantization Framework for Accurate and Certifiably Robust DNNs","date":"2024-10-31","arxiv_id":"2410.24214","repositories_listed":0,"syntology":null},{"url":null,"slug":"breaking-determinism-fuzzy-modeling-of","title":"Breaking Determinism: Fuzzy Modeling of Sequential Recommendation Using Discrete State Space Diffusion Model","date":"2024-10-31","arxiv_id":"2410.23994","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-comprehensive-study-on-quantization","title":"A Comprehensive Study on Quantization Techniques for Large Language Models","date":"2024-10-30","arxiv_id":"2411.02530","repositories_listed":0,"syntology":null},{"url":null,"slug":"accelerated-ai-inference-via-dynamic","title":"Accelerated AI Inference via Dynamic Execution Methods","date":"2024-10-30","arxiv_id":"2411.00853","repositories_listed":0,"syntology":null},{"url":null,"slug":"apcodec-a-spectrum-coding-based-high-fidelity","title":"APCodec+: A Spectrum-Coding-Based High-Fidelity and High-Compression-Rate Neural Audio Codec with Staged Training Paradigm","date":"2024-10-30","arxiv_id":"2410.22807","repositories_listed":0,"syntology":null},{"url":null,"slug":"elmgs-enhancing-memory-and-computation","title":"ELMGS: Enhancing memory and computation scaLability through coMpression for 3D Gaussian Splatting","date":"2024-10-30","arxiv_id":"2410.23213","repositories_listed":0,"syntology":null},{"url":null,"slug":"gwq-gradient-aware-weight-quantization-for","title":"GWQ: Gradient-Aware Weight Quantization for Large Language Models","date":"2024-10-30","arxiv_id":"2411.00850","repositories_listed":0,"syntology":null},{"url":null,"slug":"hrpvt-high-resolution-pyramid-vision","title":"HRPVT: High-Resolution Pyramid Vision Transformer for medium and small-scale human pose estimation","date":"2024-10-29","arxiv_id":"2410.22079","repositories_listed":0,"syntology":null},{"url":"/paper/eora-training-free-compensation-for","slug":"eora-training-free-compensation-for","title":"EoRA: Training-free Compensation for Compressed LLM with Eigenspace Low-Rank Approximation","date":"2024-10-28","arxiv_id":"2410.21271","repositories_listed":0,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/eora-training-free-compensation-for#ran","syntology_url":"https://syntology.ai/paper/2410.21271","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21271"}},"official":null}},{"url":null,"slug":"logarithmically-quantized-distributed","title":"Logarithmically Quantized Distributed Optimization over Dynamic Multi-Agent Networks","date":"2024-10-27","arxiv_id":"2410.20345","repositories_listed":0,"syntology":null},{"url":null,"slug":"unleashing-dynamic-range-and-resolution-in","title":"Unleashing Dynamic Range and Resolution in Unlimited Sensing Framework via Novel Hardware","date":"2024-10-26","arxiv_id":"2410.20193","repositories_listed":0,"syntology":null},{"url":null,"slug":"you-never-know-quantization-induces","title":"You Never Know: Quantization Induces Inconsistent Biases in Vision-Language Foundation Models","date":"2024-10-26","arxiv_id":"2410.20265","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-survey-of-small-language-models","title":"A Survey of Small Language Models","date":"2024-10-25","arxiv_id":"2410.20011","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-id-free-item-representation-with","title":"Learning ID-free Item Representation with Token Crossing for Multimodal Recommendation","date":"2024-10-25","arxiv_id":"2410.19276","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-counterexample-in-cross-correlation","title":"A Counterexample in Cross-Correlation Template Matching","date":"2024-10-24","arxiv_id":"2410.19085","repositories_listed":0,"syntology":null},{"url":null,"slug":"sliding-dft-based-signal-recovery-for-modulo","title":"Sliding DFT-based Signal Recovery for Modulo ADC with 1-bit Folding Information","date":"2024-10-24","arxiv_id":"2410.18757","repositories_listed":0,"syntology":null},{"url":null,"slug":"tesseraq-ultra-low-bit-llm-post-training","title":"TesseraQ: Ultra Low-Bit LLM Post-Training Quantization with Block Reconstruction","date":"2024-10-24","arxiv_id":"2410.19103","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-nature-of-mathematical-modeling-and","title":"The Nature of Mathematical Modeling and Probabilistic Optimization Engineering in Generative AI","date":"2024-10-24","arxiv_id":"2410.18441","repositories_listed":0,"syntology":null},{"url":null,"slug":"adaptive-wireless-image-semantic-transmission-1","title":"Adaptive Wireless Image Semantic Transmission: Design, Simulation, and Prototype Validation","date":"2024-10-23","arxiv_id":"2410.17536","repositories_listed":0,"syntology":null}],"record_sha256":"e026a27981e7d154a4c3812a57e3028a1490686a68c5cd57d016cd665eb59afd","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}