@misc{indiciaef43ed02219a7, title = {SOLO: A Single Transformer for Scalable Vision-Language Modeling}, author = {Yangyi Chen and Xingyao Wang and Hao Peng and Heng Ji}, year = {2024}, url = {https://arxiv.org/abs/2407.06438}, note = {Source identifier: 2407.06438} }