@misc{indiciae8e3b9910882f, title = {ViTextVQA: A Large-Scale Visual Question Answering Dataset and a Novel Multimodal Feature Fusion Method for Vietnamese Text Comprehension in Images}, author = {Quan Van Nguyen and Dan Quang Tran and Huy Quang Pham and Thang Kien-Bao Nguyen and Nghia Hieu Nguyen and Kiet Van Nguyen and Ngan Luu-Thuy Nguyen}, year = {2026}, doi = {10.1016/j.eswa.2025.130839}, url = {https://arxiv.org/abs/2404.10652}, note = {Source identifier: 2404.10652} }