@misc{indiciae63de78f1a389, title = {Building a Multi-modal Spatiotemporal Expert for Zero-shot Action Recognition with CLIP}, author = {Yating Yu and Congqi Cao and Yueran Zhang and Qinyi Lv and Lingtong Min and Yanning Zhang}, year = {2025}, url = {https://arxiv.org/abs/2412.09895}, note = {Source identifier: 2412.09895} }