@misc{indiciae900d87a506ba, title = {Automatic Pseudo-Harmful Prompt Generation for Evaluating False Refusals in Large Language Models}, author = {Bang An and Sicheng Zhu and Ruiyi Zhang and Michael-Andrei Panaitescu-Liess and Yuancheng Xu and Furong Huang}, year = {2025}, url = {https://arxiv.org/abs/2409.00598}, note = {Source identifier: 2409.00598} }