@misc{indiciae3476fee8a797, title = {Effective End-to-End Vision Language Pretraining with Semantic Visual Loss}, author = {Xiaofeng Yang and Fayao Liu and Guosheng Lin}, year = {2023}, doi = {10.1109/tmm.2023.3237166}, url = {https://arxiv.org/abs/2301.07236}, note = {Source identifier: 2301.07236} }