@misc{indiciae90ce4240a3cc, title = {Beyond Penalizing Mistakes: Stabilizing Efficiency Training in Large Reasoning Models via Adaptive Correct-Only Rewards}, author = {Jungseob Lee and Seungyoon Lee and Seongtae Hong and Minhyuk Kim and Chanjun Park and Heuiseok Lim}, year = {2026}, url = {https://arxiv.org/abs/2606.22716}, note = {Source identifier: 2606.22716} }