@misc{indiciaeb1b722b67660, title = {Can LLMs Deceive CLIP? Benchmarking Adversarial Compositionality of Pre-trained Multimodal Representation via Text Updates}, author = {Jaewoo Ahn and Heeseung Yun and Dayoon Ko and Gunhee Kim}, year = {2025}, url = {https://arxiv.org/abs/2505.22943}, note = {Source identifier: 2505.22943} }