@misc{indiciaec4de2c734f5c, title = {Stress-testing medical large language models reveals latent safety pathology beyond benchmark accuracy}, author = {Yuan Shen and Xiaojun Wu and Linghua Yu}, year = {2026}, url = {https://arxiv.org/abs/2606.07929}, note = {Source identifier: 2606.07929} }