@misc{indiciaefd41240c0075, title = {VISTA: A Visual and Textual Attention Dataset for Interpreting Multimodal Models}, author = {Harshit and Tolga Tasdizen}, year = {2024}, url = {https://arxiv.org/abs/2410.04609}, note = {Source identifier: 2410.04609} }