{"url":"/method/crossbow","slug":"crossbow","name":"Crossbow","full_name":"Crossbow","full_name_withheld":false,"description_markdown":"**Crossbow** is a single-server multi-GPU system for training deep learning models that enables users to freely choose their preferred batch size—however small—while scaling to multiple GPUs. Crossbow uses many parallel model replicas and avoids reduced statistical efficiency through a new synchronous training method. [SMA](https://paperswithcode.com/method/slime-mould-algorithm-sma), a synchronous variant of model averaging, is used in which replicas independently explore the solution space with gradient descent, but adjust their search synchronously based on the trajectory of a globally-consistent average model.","description_state":"present","introduced_year":null,"introduced_by":{"title":"CROSSBOW: Scaling Deep Learning with Small Batch Sizes on Multi-GPU Servers","paper":"/paper/crossbow-scaling-deep-learning-with-small","first_author":"Alexandros Koliousis","n_authors":6,"url_abs":null,"archive_paper_url":"https://paperswithcode.com/paper/crossbow-scaling-deep-learning-with-small"},"source":{"url":"http://arxiv.org/abs/1901.02244v1","title":"CROSSBOW: Scaling Deep Learning with Small Batch Sizes on Multi-GPU Servers","url_on_a_paper_host":true},"code_snippet_url":null,"code_snippet_url_on_a_code_host":false,"categories":[{"area":"General","area_id":"general","collection":"Asynchronous Data Parallel","url":"/methods/category/asynchronous-data-parallel","pwc_aliases":[]},{"area":"General","area_id":"general","collection":"Data Parallel Methods","url":"/methods/category/data-parallel-methods","pwc_aliases":[]},{"area":"General","area_id":"general","collection":"Distributed Methods","url":"/methods/category/distributed-methods","pwc_aliases":[]}],"n_papers_tagged":2,"archive_num_papers":2,"papers_newest_first":[{"paper":null,"title":"A Low-Delay MAC for IoT Applications: Decentralized Optimal Scheduling of Queues without Explicit State Information Sharing","date":"2021-05-24","arxiv_id":"2105.11213","n_code_links":0,"syntology":null},{"paper":"/paper/crossbow-scaling-deep-learning-with-small","title":"CROSSBOW: Scaling Deep Learning with Small Batch Sizes on Multi-GPU Servers","date":"2019-01-08","arxiv_id":"1901.02244","n_code_links":1,"syntology":null}],"papers_shown":2,"tasks":[{"task":"/task/deep-learning","name":"Deep Learning","papers":1},{"task":"/task/fairness","name":"Fairness","papers":1},{"task":null,"name":"GPU","papers":1},{"task":"/task/scheduling","name":"Scheduling","papers":1}],"tasks_shown":4,"n_tasks":4,"usage_by_year":[{"year":"2019","papers":1},{"year":"2021","papers":1}],"row_source":"methods_table","archive":{"source":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","archive_url":"https://paperswithcode.com/method/crossbow"},"syntology_read_at":"2026-09-24T18:15:14+00:00"}