@misc{indiciae432fbc091f6a, title = {Tapered Off-Policy REINFORCE: Stable and efficient reinforcement learning for LLMs}, author = {Nicolas Le Roux and Marc G. Bellemare and Jonathan Lebensold and Arnaud Bergeron and Joshua Greaves and Alex Fréchette and Carolyne Pelletier and Eric Thibodeau-Laufer and Sándor Toth and Sam Work}, year = {2025}, url = {https://arxiv.org/abs/2503.14286}, note = {Source identifier: 2503.14286} }