@misc{indiciae977e910eab87, title = {LLM Safety From Within: Detecting Harmful Content with Internal Representations}, author = {Difan Jiao and Yilun Liu and Ye Yuan and Zhenwei Tang and Linfeng Du and Haolun Wu and Ashton Anderson}, year = {2026}, doi = {10.18653/v1/2026.acl-long.1844}, url = {https://arxiv.org/abs/2604.18519}, note = {Source identifier: 2604.18519} }