@misc{indiciaea61d0507ec48, title = {VICTR: Visual Information Captured Text Representation for Text-to-Image Multimodal Tasks}, author = {Soyeon Caren Han and Siqu Long and Siwen Luo and Kunze Wang and Josiah Poon}, year = {2020}, url = {https://arxiv.org/abs/2010.03182}, note = {Source identifier: 2010.03182} }