@misc{indiciae5254df03158a, title = {GRPO and Reflection Reward for Mathematical Reasoning in Large Language Models}, author = {Zhijie Wang}, year = {2026}, doi = {10.54254/2755-2721/2025.tj23144}, url = {https://arxiv.org/abs/2603.14041}, note = {Source identifier: 2603.14041} }