@misc{indiciae0762722e8442, title = {Self-Captioning Multimodal Interaction Tuning: Amplifying Exploitable Redundancies for Robust Vision Language Models}, author = {Yuriel Ryan and Hei Man Ip and Adriel Kuek and Paul Pu Liang and Roy Ka-Wei Lee}, year = {2026}, url = {https://arxiv.org/abs/2605.08145}, note = {Source identifier: 2605.08145} }