@misc{indiciaeb7abf6b7eece, title = {Reason in Chains, Learn in Trees: Self-Rectification and Grafting for Multi-turn Agent Policy Optimization}, author = {Yu Li and Sizhe Tang and Tian Lan}, year = {2026}, url = {https://arxiv.org/abs/2604.07165}, note = {Source identifier: 2604.07165} }