@misc{indiciaecf80712808c3, title = {ManagerTower: Aggregating the Insights of Uni-Modal Experts for Vision-Language Representation Learning}, author = {Xiao Xu and Bei Li and Chenfei Wu and Shao-Yen Tseng and Anahita Bhiwandiwalla and Shachar Rosenman and Vasudev Lal and Wanxiang Che and Nan Duan}, year = {2023}, url = {https://arxiv.org/abs/2306.00103}, note = {Source identifier: 2306.00103} }