@misc{indiciaec9e4fdd6e0c2, title = {Rethinking RL for LLM Reasoning: It's Sparse Policy Selection, Not Capability Learning}, author = {Ömer Faruk Akgül and Rajgopal Kannan and Willie Neiswanger and Viktor Prasanna}, year = {2026}, url = {https://arxiv.org/abs/2605.06241}, note = {Source identifier: 2605.06241} }