@misc{indiciaeab961928defe, title = {RLSR: Reinforcement Learning with Supervised Reward Outperforms SFT in Instruction Following}, author = {Zhichao Wang and Andy Wong and Ruslan Belkin}, year = {2025}, url = {https://arxiv.org/abs/2510.14200}, note = {Source identifier: 2510.14200} }