@misc{indiciaee908e3e8467e, title = {Towards a Unified Foundation Model: Jointly Pre-Training Transformers on Unpaired Images and Text}, author = {Qing Li and Boqing Gong and Yin Cui and Dan Kondratyuk and Xianzhi Du and Ming-Hsuan Yang and Matthew Brown}, year = {2021}, url = {https://arxiv.org/abs/2112.07074}, note = {Source identifier: 2112.07074} }