@misc{indiciae870633c3b6c3, title = {Continual Pre-Training of Large Language Models: How to (re)warm your model?}, author = {Kshitij Gupta and Benjamin Thérien and Adam Ibrahim and Mats L. Richter and Quentin Anthony and Eugene Belilovsky and Irina Rish and Timothée Lesort}, year = {2023}, url = {https://arxiv.org/abs/2308.04014}, note = {Source identifier: 2308.04014} }