@misc{indiciaeae62c9aef194, title = {ViCLIP-OT: The First Foundation Vision-Language Model for Vietnamese Image-Text Retrieval with Optimal Transport}, author = {Quoc-Khang Tran and Minh-Thien Nguyen and Nguyen-Khang Pham}, year = {2026}, url = {https://arxiv.org/abs/2602.22678}, note = {Source identifier: 2602.22678} }