@misc{indiciae7f102c4bb405, title = {Why Self-Rewarding Works: Theoretical Guarantees for Iterative Alignment of Language Models}, author = {Shi Fu and Yingjie Wang and Shengchao Hu and Peng Wang and Dacheng Tao}, year = {2026}, url = {https://arxiv.org/abs/2601.22513}, note = {Source identifier: 2601.22513} }