@misc{indiciae676c519c906f, title = {Beyond Trajectory Imitation: Strategy-Guided Policy Optimization for LLM Reasoning}, author = {Tianyuan Shi and Canbin Huang and Bei Li and Xin Chen and Xiaojun Quan and Jingang Wang and Qifan Wang}, year = {2026}, url = {https://arxiv.org/abs/2606.24064}, note = {Source identifier: 2606.24064} }