@misc{indiciae343d2c656d39, title = {Response-Level Rewards Are All You Need for Online Reinforcement Learning in LLMs: A Mathematical Perspective}, author = {Shenghua He and Tian Xia and Xuan Zhou and Hui Wei}, year = {2025}, url = {https://arxiv.org/abs/2506.02553}, note = {Source identifier: 2506.02553} }