@misc{indiciaefc612082b492, title = {Rethinking Reinforcement fine-tuning of LLMs: A Multi-armed Bandit Learning Perspective}, author = {Xiao Hu and Hong Xie and Tao Tan and Defu Lian and Jianyu Han}, year = {2026}, url = {https://arxiv.org/abs/2601.14599}, note = {Source identifier: 2601.14599} }