@misc{indiciaebee96e8a3e96, title = {VLMT: Vision-Language Multimodal Transformer for Multimodal Multi-hop Question Answering}, author = {Qi Zhi Lim and Chin Poo Lee and Kian Ming Lim and Kalaiarasi Sonai Muthu Anbananthen}, year = {2025}, url = {https://arxiv.org/abs/2504.08269}, note = {Source identifier: 2504.08269} }