@misc{indiciae50864339784a, title = {E-ViLM: Efficient Video-Language Model via Masked Video Modeling with Semantic Vector-Quantized Tokenizer}, author = {Jacob Zhiyuan Fang and Skyler Zheng and Vasu Sharma and Robinson Piramuthu}, year = {2023}, url = {https://arxiv.org/abs/2311.17267}, note = {Source identifier: 2311.17267} }