@misc{indiciae46b7ec32ae4c, title = {Layered gradient accumulation and modular pipeline parallelism: fast and efficient training of large language models}, author = {Joel Lamy-Poirier}, year = {2021}, url = {https://arxiv.org/abs/2106.02679}, note = {Source identifier: 2106.02679} }