@misc{indiciae312a6712ca4d, title = {Simple and Scalable Strategies to Continually Pre-train Large Language Models}, author = {Adam Ibrahim and Benjamin Thérien and Kshitij Gupta and Mats L. Richter and Quentin Anthony and Timothée Lesort and Eugene Belilovsky and Irina Rish}, year = {2024}, url = {https://arxiv.org/abs/2403.08763}, note = {Source identifier: 2403.08763} }