@misc{indiciaef9e135d43617, title = {Parallel Restarted SGD with Faster Convergence and Less Communication: Demystifying Why Model Averaging Works for Deep Learning}, author = {Hao Yu and Sen Yang and Shenghuo Zhu}, year = {2018}, url = {https://arxiv.org/abs/1807.06629}, note = {Source identifier: 1807.06629} }