@misc{indiciaea16850282987, title = {Demystifying Design Choices of Reinforcement Fine-tuning: A Batched Contextual Bandit Learning Perspective}, author = {Hong Xie and Xiao Hu and Tao Tan and Haoran Gu and Xin Li and Jianyu Han and Defu Lian and Enhong Chen}, year = {2026}, url = {https://arxiv.org/abs/2601.22532}, note = {Source identifier: 2601.22532} }