@misc{indiciae3478324dccb9, title = {ByteScale: Efficient Scaling of LLM Training with a 2048K Context Length on More Than 12,000 GPUs}, author = {Hao Ge and Junda Feng and Qi Huang and Fangcheng Fu and Xiaonan Nie and Lei Zuo and Haibin Lin and Bin Cui and Xin Liu}, year = {2025}, doi = {10.1145/3718958.3754352}, url = {https://arxiv.org/abs/2502.21231}, note = {Source identifier: 2502.21231} }