@misc{indiciaec797dc56ea50, title = {Latent Adversarial Training Improves Robustness to Persistent Harmful Behaviors in LLMs}, author = {Abhay Sheshadri and Aidan Ewart and Phillip Guo and Aengus Lynch and Cindy Wu and Vivek Hebbar and Henry Sleight and Asa Cooper Stickland and Ethan Perez and Dylan Hadfield-Menell and Stephen Casper}, year = {2025}, url = {https://arxiv.org/abs/2407.15549}, note = {Source identifier: 2407.15549} }