@misc{indiciae5612fef9e1d5, title = {"Moralized" Multi-Step Jailbreak Prompts: Black-Box Testing of Guardrails in Large Language Models for Verbal Attacks}, author = {Libo Wang}, year = {2025}, url = {https://arxiv.org/abs/2411.16730}, note = {Source identifier: 2411.16730} }