@misc{indiciae8c5b257fae88, title = {TIC-GRPO: Provable and Efficient Optimization for Reinforcement Learning from Human Feedback}, author = {Lei Pang and Jun Luo and Ruinan Jin}, year = {2026}, url = {https://arxiv.org/abs/2508.02833}, note = {Source identifier: 2508.02833} }