@misc{indiciae6c7f64dccaa8, title = {Beyond Text: Multimodal Jailbreaking of Vision-Language and Audio Models through Perceptually Simple Transformations}, author = {Divyanshu Kumar and Shreyas Jena and Nitin Aravind Birur and Tanay Baswa and Sahil Agarwal and Prashanth Harshangi}, year = {2025}, url = {https://arxiv.org/abs/2510.20223}, note = {Source identifier: 2510.20223} }