@misc{indiciaeb6568f2d17ba, title = {VideoINSTA: Zero-shot Long Video Understanding via Informative Spatial-Temporal Reasoning with LLMs}, author = {Ruotong Liao and Max Erler and Huiyu Wang and Guangyao Zhai and Gengyuan Zhang and Yunpu Ma and Volker Tresp}, year = {2024}, url = {https://arxiv.org/abs/2409.20365}, note = {Source identifier: 2409.20365} }