@misc{indiciaed516aa9252f5, title = {Do Vision \& Language Decoders use Images and Text equally? How Self-consistent are their Explanations?}, author = {Letitia Parcalabescu and Anette Frank}, year = {2025}, url = {https://arxiv.org/abs/2404.18624}, note = {Source identifier: 2404.18624} }