@misc{indiciae2d0561ac76dc, title = {An Expectation-Maximization Perspective on Reinforcement Learning for LLM Reasoning}, author = {Tianbing Xu}, year = {2026}, url = {https://arxiv.org/abs/2504.18587}, note = {Source identifier: 2504.18587} }