@misc{indiciaeb47d11527bea, title = {Scaling Vision-Language Models with Sparse Mixture of Experts}, author = {Sheng Shen and Zhewei Yao and Chunyuan Li and Trevor Darrell and Kurt Keutzer and Yuxiong He}, year = {2023}, url = {https://arxiv.org/abs/2303.07226}, note = {Source identifier: 2303.07226} }