@misc{indiciae2ce2b9156e9d, title = {UniTAB: Unifying Text and Box Outputs for Grounded Vision-Language Modeling}, author = {Zhengyuan Yang and Zhe Gan and Jianfeng Wang and Xiaowei Hu and Faisal Ahmed and Zicheng Liu and Yumao Lu and Lijuan Wang}, year = {2022}, url = {https://arxiv.org/abs/2111.12085}, note = {Source identifier: 2111.12085} }