@misc{indiciae9141a7976c7c, title = {Jackpot: Optimal Budgeted Rejection Sampling for Extreme Actor-Policy Mismatch Reinforcement Learning}, author = {Zhuoming Chen and Hongyi Liu and Yang Zhou and Haizhong Zheng and Beidi Chen}, year = {2026}, url = {https://arxiv.org/abs/2602.06107}, note = {Source identifier: 2602.06107} }