@misc{indiciae975bbe008cf5, title = {STEP: Staged Parameter-Efficient Pre-training for Large Language Models}, author = {Kazuki Yano and Takumi Ito and Jun Suzuki}, year = {2025}, url = {https://arxiv.org/abs/2504.04151}, note = {Source identifier: 2504.04151} }