@misc{indiciae659cf079e737, title = {Scaling Pre-training to One Hundred Billion Data for Vision Language Models}, author = {Xiao Wang and Ibrahim Alabdulmohsin and Daniel Salz and Zhe Li and Keran Rong and Xiaohua Zhai}, year = {2026}, url = {https://arxiv.org/abs/2502.07617}, note = {Source identifier: 2502.07617} }