@misc{indiciae203e2d3283e0, title = {Detecting and Understanding Vulnerabilities in Language Models via Mechanistic Interpretability}, author = {Jorge García-Carrasco and Alejandro Maté and Juan Trujillo}, year = {2024}, doi = {10.24963/ijcai.2024/43}, url = {https://arxiv.org/abs/2407.19842}, note = {Source identifier: 2407.19842} }