@misc{indiciae0a324e65d598, title = {FashionViL: Fashion-Focused Vision-and-Language Representation Learning}, author = {Xiao Han and Licheng Yu and Xiatian Zhu and Li Zhang and Yi-Zhe Song and Tao Xiang}, year = {2022}, url = {https://arxiv.org/abs/2207.08150}, note = {Source identifier: 2207.08150} }