@misc{indiciae9fe810d54fa4, title = {Trust the Batch, On- or Off-Policy: Adaptive Policy Optimization for RL Post-Training}, author = {Rasool Fakoor and Murdock Aubry and Nicholas Stranges and Alexander J. Smola}, year = {2026}, url = {https://arxiv.org/abs/2605.12380}, note = {Source identifier: 2605.12380} }