@misc{indiciaee3d9f4f47a74, title = {Chat-UniVi: Unified Visual Representation Empowers Large Language Models with Image and Video Understanding}, author = {Peng Jin and Ryuichi Takanobu and Wancai Zhang and Xiaochun Cao and Li Yuan}, year = {2024}, url = {https://arxiv.org/abs/2311.08046}, note = {Source identifier: 2311.08046} }