{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/post-training-4-bit-quantization-of","title":"Post-training 4-bit quantization of convolution networks for rapid-deployment","arxiv_id":"1810.05723","date":"2018-10-02","proceeding":null,"authors":["Ron Banner","Yury Nahshan","Elad Hoffer","Daniel Soudry"],"abstract":"Convolutional neural networks require significant memory bandwidth and storage for intermediate computations, apart from substantial computing resources. Neural network quantization has significant benefits in reducing the amount of intermediate results, but it often requires the full datasets and time-consuming fine tuning to recover the accuracy lost after quantization. This paper introduces the first practical 4-bit post training quantization approach: it does not involve training the quantized model (fine-tuning), nor it requires the availability of the full dataset. We target the quantization of both activations and weights and suggest three complementary methods for minimizing quantization error at the tensor level, two of whom obtain a closed-form analytical solution. Combining these methods, our approach achieves accuracy that is just a few percents less the state-of-the-art baseline across a wide range of convolutional models. The source code to replicate all experiments is available on GitHub: \\url{https://github.com/submission2019/cnn-quantization}.","url_abs":"https://arxiv.org/abs/1810.05723v3","url_pdf":"https://arxiv.org/pdf/1810.05723v3.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"post-training-4-bit-quantization-of","repo_url":"https://github.com/submission2019/cnn-quantization","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":0,"framework":"pytorch","reach":null},{"paper_slug":"post-training-4-bit-quantization-of","repo_url":"https://github.com/AkashB23/4-bit-quantization-with-tensorflow-1.15.2","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":{"status":"ok"}}],"tasks":[{"task_slug":"quantization","task_name":"Quantization"}],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1810.05723","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"1810.05723"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/submission2019/cnn-quantization","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/AkashB23/4-bit-quantization-with-tensorflow-1.15.2","reach":{"status":"ok"}}],"summary":{"ran_violates":1,"ran_draft_wrong":1,"ran_honours":1,"unverified":1},"by_repo_kind":{"official":{"samples":4,"ran":3,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":4,"samples":[{"code_sha256_prefix":"03518893ebde1f8f","entry":"laplace_prior_mse","repo":"submission2019/cnn-quantization","repo_kind":"official","path":"pytorch_quantizer/quantization/qtypes/int_quantizer.py","file_url":"https://github.com/submission2019/cnn-quantization/blob/HEAD/pytorch_quantizer/quantization/qtypes/int_quantizer.py","link_basis":"first_harvest_node","language":"python","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"03518893ebde1f8f"}},{"code_sha256_prefix":"51355bfea2e7e3a2","entry":"to_cuda","repo":"submission2019/cnn-quantization","repo_kind":"official","path":"pytorch_quantizer/quantization/qtypes/int_quantizer.py","file_url":"https://github.com/submission2019/cnn-quantization/blob/HEAD/pytorch_quantizer/quantization/qtypes/int_quantizer.py","link_basis":"first_harvest_node","language":"python","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"51355bfea2e7e3a2"}},{"code_sha256_prefix":"2a066b9a11c2f824","entry":"to_numpy","repo":"submission2019/cnn-quantization","repo_kind":"official","path":"pytorch_quantizer/quantization/qtypes/int_quantizer.py","file_url":"https://github.com/submission2019/cnn-quantization/blob/HEAD/pytorch_quantizer/quantization/qtypes/int_quantizer.py","link_basis":"first_harvest_node","language":"python","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"2a066b9a11c2f824"}},{"code_sha256_prefix":"f7e52a87dc81b79c","entry":"get_params","repo":"submission2019/cnn-quantization","repo_kind":"official","path":"inference/inference_sim.py","file_url":"https://github.com/submission2019/cnn-quantization/blob/HEAD/inference/inference_sim.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"mcp_get_code":{"code_sha256":"f7e52a87dc81b79c"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}