@misc{indiciaea424ff255733, title = {VideoAgent: Long-form Video Understanding with Large Language Model as Agent}, author = {Xiaohan Wang and Yuhui Zhang and Orr Zohar and Serena Yeung-Levy}, year = {2024}, url = {https://arxiv.org/abs/2403.10517}, note = {Source identifier: 2403.10517} }