@misc{indiciaeda59234d9b92, title = {Weak-to-Strong Preference Optimization: Stealing Reward from Weak Aligned Model}, author = {Wenhong Zhu and Zhiwei He and Xiaofeng Wang and Pengfei Liu and Rui Wang}, year = {2025}, url = {https://arxiv.org/abs/2410.18640}, note = {Source identifier: 2410.18640} }