@misc{indiciaedb1a14db1926, title = {Towards Explainable AI: Multi-Modal Transformer for Video-based Image Description Generation}, author = {Lakshita Agarwal and Bindu Verma}, year = {2025}, doi = {10.1007/s11760-026-05233-5}, url = {https://arxiv.org/abs/2504.16788}, note = {Source identifier: 2504.16788} }