{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/training-a-vision-transformer-from-scratch-in","title":"Training a Vision Transformer from scratch in less than 24 hours with 1 GPU","arxiv_id":"2211.05187","date":"2022-11-09","proceeding":null,"authors":["Saghar Irandoust","Thibaut Durand","Yunduz Rakhmangulova","Wenjie Zi","Hossein Hajimirsadeghi"],"abstract":"Transformers have become central to recent advances in computer vision. However, training a vision Transformer (ViT) model from scratch can be resource intensive and time consuming. In this paper, we aim to explore approaches to reduce the training costs of ViT models. We introduce some algorithmic improvements to enable training a ViT model from scratch with limited hardware (1 GPU) and time (24 hours) resources. First, we propose an efficient approach to add locality to the ViT architecture. Second, we develop a new image size curriculum learning strategy, which allows to reduce the number of patches extracted from each image at the beginning of the training. Finally, we propose a new variant of the popular ImageNet1k benchmark by adding hardware and time constraints. We evaluate our contributions on this benchmark, and show they can significantly improve performances given the proposed training budget. We will share the code in https://github.com/BorealisAI/efficient-vit-training.","url_abs":"https://arxiv.org/abs/2211.05187v1","url_pdf":"https://arxiv.org/pdf/2211.05187v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"training-a-vision-transformer-from-scratch-in","repo_url":"https://github.com/Abdulrahman-Adel/Real-Life-Violence-Detection","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"tf","reach":null}],"tasks":[{"task_slug":null,"task_name":"GPU"}],"methods":[{"method_slug":"absolute-position-encodings","method_name":"Absolute Position Encodings"},{"method_slug":"adam","method_name":"Adam"},{"method_slug":"attention","method_name":"Attention"},{"method_slug":"bpe","method_name":"BPE"},{"method_slug":"dense-connections","method_name":"Dense Connections"},{"method_slug":"dropout","method_name":"Dropout"},{"method_slug":"label-smoothing","method_name":"Label Smoothing"},{"method_slug":"layer-normalization","method_name":"Layer Normalization"},{"method_slug":"linear-layer","method_name":"Linear Layer"},{"method_slug":"multi-head-attention","method_name":"Multi-Head Attention"},{"method_slug":"position-wise-feed-forward-layer","method_name":"Position-Wise Feed-Forward Layer"},{"method_slug":"residual-connection","method_name":"Residual Connection"},{"method_slug":"softmax","method_name":"Softmax"},{"method_slug":"transformer","method_name":"Transformer"},{"method_slug":"vision-transformer","method_name":"Vision Transformer"}],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2211.05187","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2211.05187"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"deterministic:regex_extraction","url":"https://github.com/BorealisAI/efficient-vit-training","reach":null},{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/Abdulrahman-Adel/Real-Life-Violence-Detection","reach":null}],"summary":{"ran":4,"unverified":2},"by_repo_kind":{"found_in_text":{"samples":6,"ran":4,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":6,"samples":[{"code_sha256_prefix":"914feb6bd11b62e4","entry":"Attention","repo":"BorealisAI/efficient-vit-training","repo_kind":"found_in_text","path":"models_vit/localvit.py","file_url":"https://github.com/BorealisAI/efficient-vit-training/blob/HEAD/models_vit/localvit.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"914feb6bd11b62e4"}},{"code_sha256_prefix":"d1cef517b7b6e6c5","entry":"ECALayer","repo":"BorealisAI/efficient-vit-training","repo_kind":"found_in_text","path":"models_vit/localvit.py","file_url":"https://github.com/BorealisAI/efficient-vit-training/blob/HEAD/models_vit/localvit.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"d1cef517b7b6e6c5"}},{"code_sha256_prefix":"4661b2978d600396","entry":"LocalityFeedForward","repo":"BorealisAI/efficient-vit-training","repo_kind":"found_in_text","path":"models_vit/localvit.py","file_url":"https://github.com/BorealisAI/efficient-vit-training/blob/HEAD/models_vit/localvit.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"4661b2978d600396"}},{"code_sha256_prefix":"b0b56eb3a5da1555","entry":"SELayer","repo":"BorealisAI/efficient-vit-training","repo_kind":"found_in_text","path":"models_vit/localvit.py","file_url":"https://github.com/BorealisAI/efficient-vit-training/blob/HEAD/models_vit/localvit.py","link_basis":"first_harvest_node","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"b0b56eb3a5da1555"}},{"code_sha256_prefix":"f4849c8a28f22c7b","entry":"Block","repo":"BorealisAI/efficient-vit-training","repo_kind":"found_in_text","path":"models_vit/localvit.py","file_url":"https://github.com/BorealisAI/efficient-vit-training/blob/HEAD/models_vit/localvit.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"f4849c8a28f22c7b"}},{"code_sha256_prefix":"8727b4d2f1f91140","entry":"LocalVisionTransformer","repo":"BorealisAI/efficient-vit-training","repo_kind":"found_in_text","path":"models_vit/localvit.py","file_url":"https://github.com/BorealisAI/efficient-vit-training/blob/HEAD/models_vit/localvit.py","link_basis":"first_harvest_node","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"mcp_get_code":{"code_sha256":"8727b4d2f1f91140"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}