@misc{indiciaec39e08c0f2eb, title = {Robust Harmful Features Under Jailbreak Attacks: Mechanistic Evidence from Attention Head Specialization in Large Language Models}, author = {Yanchen Yin and Dongqi Han and Linghui Li}, year = {2026}, url = {https://arxiv.org/abs/2606.28153}, note = {Source identifier: 2606.28153} }