@misc{indiciae107f1d87999d, title = {Why Multi-Step Tool-Use Reinforcement Learning Collapses and How Supervisory Signals Fix It}, author = {Yupu Hao and Zhuoran Jin and Huanxuan Liao and Kang Liu and Jun Zhao}, year = {2026}, url = {https://arxiv.org/abs/2606.26027}, note = {Source identifier: 2606.26027} }