@misc{indiciae3b686dd3a67d, title = {Leash: Adaptive Length Penalty and Reward Shaping for Efficient Large Reasoning Model}, author = {Yanhao Li and Lu Ma and Jiaran Zhang and Lexiang Tang and Wentao Zhang and Guibo Luo}, year = {2025}, url = {https://arxiv.org/abs/2512.21540}, note = {Source identifier: 2512.21540} }