@misc{indiciae6907e7269f5b, title = {VISTA: Enhancing Long-Duration and High-Resolution Video Understanding by Video Spatiotemporal Augmentation}, author = {Weiming Ren and Huan Yang and Jie Min and Cong Wei and Wenhu Chen}, year = {2024}, url = {https://arxiv.org/abs/2412.00927}, note = {Source identifier: 2412.00927} }