{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/modeling-deep-learning-accelerator-enabled","title":"Modeling Deep Learning Accelerator Enabled GPUs","arxiv_id":"1811.08309","date":"2018-11-19","proceeding":null,"authors":["Md Aamir Raihan","Negar Goli","Tor Aamodt"],"abstract":"The efficacy of deep learning has resulted in its use in a growing number of applications. The Volta graphics processor unit (GPU) architecture from NVIDIA introduced a specialized functional unit, the \"tensor core\", that helps meet the growing demand for higher performance for deep learning. In this paper we study the design of the tensor cores in NVIDIA's Volta and Turing architectures. We further propose an architectural model for the tensor cores in Volta. When implemented a GPU simulator, GPGPU-Sim, our tensor core model achieves 99.6\\% correlation versus an NVIDIA Titan~V GPU in terms of average instructions per cycle when running tensor core enabled GEMM workloads. We also describe support added to enable GPGPU-Sim to run CUTLASS, an open-source CUDA C++ template library providing customizable GEMM templates that utilize tensor cores.","url_abs":"http://arxiv.org/abs/1811.08309v1","url_pdf":"http://arxiv.org/pdf/1811.08309v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"links_only","authors_date_abstract":"arXiv metadata, CC0 1.0 (https://info.arxiv.org/help/license), from the Kaggle arXiv metadata snapshot of 2026-09-12"},"code_links":[{"paper_slug":"modeling-deep-learning-accelerator-enabled","repo_url":"https://github.com/William-An/gpgpu-sim_distribution","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"modeling-deep-learning-accelerator-enabled","repo_url":"https://github.com/accel-sim/gpgpu-sim_distribution","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"modeling-deep-learning-accelerator-enabled","repo_url":"https://github.com/csl-iisc/ScoRD","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"modeling-deep-learning-accelerator-enabled","repo_url":"https://github.com/gpgpu-sim/gpgpu-sim_distribution","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"modeling-deep-learning-accelerator-enabled","repo_url":"https://github.com/gunjae/gpgpusim-4.0.1","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"modeling-deep-learning-accelerator-enabled","repo_url":"https://github.com/harryborison/sim_dist","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"modeling-deep-learning-accelerator-enabled","repo_url":"https://github.com/nikoguil1/gpusim_SMK","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"modeling-deep-learning-accelerator-enabled","repo_url":"https://github.com/prdalmia/gpgpu-sim-tlb","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"modeling-deep-learning-accelerator-enabled","repo_url":"https://github.com/tommychouyc/deterministic-atomic-buffering","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"modeling-deep-learning-accelerator-enabled","repo_url":"https://github.com/ubc-aamodt-group/ray-intersection-predictor","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"modeling-deep-learning-accelerator-enabled","repo_url":"https://github.com/woodun/gpgpusim_new","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"modeling-deep-learning-accelerator-enabled","repo_url":"https://github.com/woodun/my_gpgpusim4_bfloat","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null},{"paper_slug":"modeling-deep-learning-accelerator-enabled","repo_url":"https://github.com/woodun/new_gpgpusim_org","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":null}],"tasks":[],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}