@misc{indiciae38880453fdbb, title = {Enhancing Video-Language Representations with Structural Spatio-Temporal Alignment}, author = {Hao Fei and Shengqiong Wu and Meishan Zhang and Min Zhang and Tat-Seng Chua and Shuicheng Yan}, year = {2024}, doi = {10.1109/tpami.2024.3393452}, url = {https://arxiv.org/abs/2406.19255}, note = {Source identifier: 2406.19255} }