{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/training-very-deep-networks","title":"Training Very Deep Networks","arxiv_id":"1507.06228","date":"2015-07-22","proceeding":"NeurIPS 2015 12","authors":["Rupesh Kumar Srivastava","Klaus Greff","Jürgen Schmidhuber"],"abstract":"Theoretical and empirical evidence indicates that the depth of neural\nnetworks is crucial for their success. However, training becomes more difficult\nas depth increases, and training of very deep networks remains an open problem.\nHere we introduce a new architecture designed to overcome this. Our so-called\nhighway networks allow unimpeded information flow across many layers on\ninformation highways. They are inspired by Long Short-Term Memory recurrent\nnetworks and use adaptive gating units to regulate the information flow. Even\nwith hundreds of layers, highway networks can be trained directly through\nsimple gradient descent. This enables the study of extremely deep and efficient\narchitectures.","url_abs":"http://arxiv.org/abs/1507.06228v2","url_pdf":"http://arxiv.org/pdf/1507.06228v2.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"training-very-deep-networks","repo_url":"https://github.com/LiyuanLucasLiu/LM-LSTM-CRF","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"unanswered"}},{"paper_slug":"training-very-deep-networks","repo_url":"https://github.com/flukeskywalker/highway-networks","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"NOASSERTION"}},{"paper_slug":"training-very-deep-networks","repo_url":"https://github.com/yoonkim/lstm-char-cnn","is_official":0,"mentioned_in_paper":0,"mentioned_in_github":1,"framework":"torch","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"image-classification","task_name":"Image Classification"}],"methods":[{"method_slug":"branch-attention","method_name":"Branch attention"},{"method_slug":"highway-networks","method_name":"Highway networks"}],"datasets_introduced":[],"methods_introduced":[{"slug":"branch-attention","name":"Branch attention","full_name":"Branch attention"}],"results":[{"leaderboard":"/sota/image-classification-on-cifar-10","task":"Image Classification","dataset":"CIFAR-10","model":"VDN","rank_in_archive_order":181,"of":265,"metrics":{"Percentage correct":"92.4"},"uses_additional_data":false},{"leaderboard":"/sota/image-classification-on-cifar-100","task":"Image Classification","dataset":"CIFAR-100","model":"VDN","rank_in_archive_order":182,"of":211,"metrics":{"Percentage correct":"67.8"},"uses_additional_data":false},{"leaderboard":"/sota/image-classification-on-mnist","task":"Image Classification","dataset":"MNIST","model":"VDN","rank_in_archive_order":37,"of":81,"metrics":{"Percentage error":"0.5"},"uses_additional_data":false}],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=1507.06228","mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}