@misc{indiciaed8f6b45feb35, title = {Video-Language Understanding: A Survey from Model Architecture, Model Training, and Data Perspectives}, author = {Thong Nguyen and Yi Bin and Junbin Xiao and Leigang Qu and Yicong Li and Jay Zhangjie Wu and Cong-Duy Nguyen and See-Kiong Ng and Luu Anh Tuan}, year = {2026}, url = {https://arxiv.org/abs/2406.05615}, note = {Source identifier: 2406.05615} }