@misc{indiciae5dc76dbd4758, title = {Using Mechanistic Interpretability to Craft Adversarial Attacks against Large Language Models}, author = {Thomas Winninger and Boussad Addad and Katarzyna Kapusta}, year = {2026}, url = {https://arxiv.org/abs/2503.06269}, note = {Source identifier: 2503.06269} }