@misc{indiciae7117c5888aaa, title = {SAFER: Probing Safety in Reward Models with Sparse Autoencoder}, author = {Wei Shi and Ziyuan Xie and Sihang Li and Xiang Wang}, year = {2026}, url = {https://arxiv.org/abs/2507.00665}, note = {Source identifier: 2507.00665} }