{"isi":1,"language":[{"iso":"eng"}],"page":"4613-4623","_id":"6558","user_id":"c635000d-4b10-11ee-a964-aac5a93f6ac1","quality_controlled":"1","day":"01","month":"12","main_file_link":[{"url":"https://arxiv.org/abs/1803.08917","open_access":"1"}],"date_updated":"2023-09-19T15:12:45Z","publication_status":"published","department":[{"_id":"DaAl"}],"article_processing_charge":"No","external_id":{"isi":["000461823304061"],"arxiv":["1803.08917"]},"title":"Byzantine stochastic gradient descent","conference":{"start_date":"2018-12-02","name":"NeurIPS: Conference on Neural Information Processing Systems","end_date":"2018-12-08","location":"Montreal, Canada"},"oa":1,"abstract":[{"lang":"eng","text":"This paper studies the problem of distributed stochastic optimization in an adversarial setting where, out of m machines which allegedly compute stochastic gradients every iteration, an α-fraction are Byzantine, and may behave adversarially. Our main result is a variant of stochastic gradient descent (SGD) which finds ε-approximate minimizers of convex functions in T=O~(1/ε²m+α²/ε²) iterations. In contrast, traditional mini-batch SGD needs T=O(1/ε²m) iterations, but cannot tolerate Byzantine failures. Further, we provide a lower bound showing that, up to logarithmic factors, our algorithm is information-theoretically optimal both in terms of sample complexity and time complexity."}],"date_created":"2019-06-13T08:22:37Z","author":[{"last_name":"Alistarh","full_name":"Alistarh, Dan-Adrian","orcid":"0000-0003-3650-940X","id":"4A899BFC-F248-11E8-B48F-1D18A9856A87","first_name":"Dan-Adrian"},{"last_name":"Allen-Zhu","full_name":"Allen-Zhu, Zeyuan","first_name":"Zeyuan"},{"last_name":"Li","full_name":"Li, Jerry","first_name":"Jerry"}],"intvolume":"      2018","publisher":"Neural Information Processing Systems Foundation","year":"2018","volume":2018,"date_published":"2018-12-01T00:00:00Z","citation":{"mla":"Alistarh, Dan-Adrian, et al. “Byzantine Stochastic Gradient Descent.” <i>Advances in Neural Information Processing Systems</i>, vol. 2018, Neural Information Processing Systems Foundation, 2018, pp. 4613–23.","short":"D.-A. Alistarh, Z. Allen-Zhu, J. Li, in:, Advances in Neural Information Processing Systems, Neural Information Processing Systems Foundation, 2018, pp. 4613–4623.","chicago":"Alistarh, Dan-Adrian, Zeyuan Allen-Zhu, and Jerry Li. “Byzantine Stochastic Gradient Descent.” In <i>Advances in Neural Information Processing Systems</i>, 2018:4613–23. Neural Information Processing Systems Foundation, 2018.","ieee":"D.-A. Alistarh, Z. Allen-Zhu, and J. Li, “Byzantine stochastic gradient descent,” in <i>Advances in Neural Information Processing Systems</i>, Montreal, Canada, 2018, vol. 2018, pp. 4613–4623.","ista":"Alistarh D-A, Allen-Zhu Z, Li J. 2018. Byzantine stochastic gradient descent. Advances in Neural Information Processing Systems. NeurIPS: Conference on Neural Information Processing Systems vol. 2018, 4613–4623.","ama":"Alistarh D-A, Allen-Zhu Z, Li J. Byzantine stochastic gradient descent. In: <i>Advances in Neural Information Processing Systems</i>. Vol 2018. Neural Information Processing Systems Foundation; 2018:4613-4623.","apa":"Alistarh, D.-A., Allen-Zhu, Z., &#38; Li, J. (2018). Byzantine stochastic gradient descent. In <i>Advances in Neural Information Processing Systems</i> (Vol. 2018, pp. 4613–4623). Montreal, Canada: Neural Information Processing Systems Foundation."},"type":"conference","publication":"Advances in Neural Information Processing Systems","oa_version":"Published Version","status":"public","scopus_import":"1"}