@misc{indiciae6fe3e3706e65, title = {M-GRPO: Stabilizing Self-Supervised Reinforcement Learning for Large Language Models with Momentum-Anchored Policy Optimization}, author = {Bizhe Bai and Hongming Wu and Peng Ye and Tao Chen}, year = {2025}, url = {https://arxiv.org/abs/2512.13070}, note = {Source identifier: 2512.13070} }