@misc{indiciae951a9d1ff411, title = {Pretraining A Large Language Model using Distributed GPUs: A Memory-Efficient Decentralized Paradigm}, author = {Jinrui Zhang and Chaodong Xiao and Aoqi Wu and Xindong Zhang and Lei Zhang}, year = {2026}, url = {https://arxiv.org/abs/2602.11543}, note = {Source identifier: 2602.11543} }