@misc{indiciaeeffb3cbada6e, title = {Tarsier2: Advancing Large Vision-Language Models from Detailed Video Description to Comprehensive Video Understanding}, author = {Liping Yuan and Jiawei Wang and Haomiao Sun and Yuchen Zhang and Yuan Lin}, year = {2025}, url = {https://arxiv.org/abs/2501.07888}, note = {Source identifier: 2501.07888} }