@misc{indiciaefcd8ee63fff4, title = {Random Policy Valuation is Enough for LLM Reasoning with Verifiable Rewards}, author = {Haoran He and Yuxiao Ye and Qingpeng Cai and Chen Hu and Binxing Jiao and Daxin Jiang and Ling Pan}, year = {2025}, url = {https://arxiv.org/abs/2509.24981}, note = {Source identifier: 2509.24981} }