@misc{indiciae316a25d5851f, title = {Robust Safety Classifier for Large Language Models: Adversarial Prompt Shield}, author = {Jinhwa Kim and Ali Derakhshan and Ian G. Harris}, year = {2023}, url = {https://arxiv.org/abs/2311.00172}, note = {Source identifier: 2311.00172} }