@misc{indiciaee71af8b3437b, title = {Trojan Activation Attack: Red-Teaming Large Language Models using Activation Steering for Safety-Alignment}, author = {Haoran Wang and Kai Shu}, year = {2024}, url = {https://arxiv.org/abs/2311.09433}, note = {Source identifier: 2311.09433} }