@misc{indiciae8f8a52c0a426, title = {Mid-Training with Self-Generated Data Improves Reinforcement Learning in Language Models}, author = {Aswin RRV and Jacob Dineen and Divij Handa and Mihir Parmar and Ben Zhou and Swaroop Mishra and Chitta Baral}, year = {2026}, url = {https://arxiv.org/abs/2605.08472}, note = {Source identifier: 2605.08472} }