@misc{indiciae2ae1d47904d1, title = {Learning Video Context as Interleaved Multimodal Sequences}, author = {Kevin Qinghong Lin and Pengchuan Zhang and Difei Gao and Xide Xia and Joya Chen and Ziteng Gao and Jinheng Xie and Xuhong Xiao and Mike Zheng Shou}, year = {2024}, url = {https://arxiv.org/abs/2407.21757}, note = {Source identifier: 2407.21757} }