@misc{indiciae3e41756c65fd, title = {VLog: Video-Language Models by Generative Retrieval of Narration Vocabulary}, author = {Kevin Qinghong Lin and Mike Zheng Shou}, year = {2025}, url = {https://arxiv.org/abs/2503.09402}, note = {Source identifier: 2503.09402} }