@misc{indiciaecf6da1fd860d, title = {Detoxifying Large Language Models via Autoregressive Reward Guided Representation Editing}, author = {Yisong Xiao and Aishan Liu and Siyuan Liang and Zonghao Ying and Xianglong Liu and Dacheng Tao}, year = {2025}, url = {https://arxiv.org/abs/2510.01243}, note = {Source identifier: 2510.01243} }