@misc{indiciaeac9f3f6d73a7, title = {GUARD: Role-playing to Generate Natural-language Jailbreakings to Test Guideline Adherence of Large Language Models}, author = {Haibo Jin and Ruoxi Chen and Peiyan Zhang and Andy Zhou and Haohan Wang}, year = {2025}, url = {https://arxiv.org/abs/2402.03299}, note = {Source identifier: 2402.03299} }