@misc{indiciaebcf7757a5c87, title = {Pixel-BERT: Aligning Image Pixels with Text by Deep Multi-Modal Transformers}, author = {Zhicheng Huang and Zhaoyang Zeng and Bei Liu and Dongmei Fu and Jianlong Fu}, year = {2020}, url = {https://arxiv.org/abs/2004.00849}, note = {Source identifier: 2004.00849} }