@misc{indiciae28c46bc66cb2, title = {MVP: Enhancing Video Large Language Models via Self-supervised Masked Video Prediction}, author = {Xiaokun Sun and Zezhong Wu and Zewen Ding and Linli Xu}, year = {2026}, url = {https://arxiv.org/abs/2601.03781}, note = {Source identifier: 2601.03781} }