@misc{indiciae3ac288d1f9a4, title = {An Interpretable N-gram Perplexity Threat Model for Large Language Model Jailbreaks}, author = {Valentyn Boreiko and Alexander Panfilov and Vaclav Voracek and Matthias Hein and Jonas Geiping}, year = {2025}, url = {https://arxiv.org/abs/2410.16222}, note = {Source identifier: 2410.16222} }