@misc{indiciae34ed0f81eb13, title = {Multimodal Transformer with Multi-View Visual Representation for Image Captioning}, author = {Jun Yu and Jing Li and Zhou Yu and Qingming Huang}, year = {2019}, url = {https://arxiv.org/abs/1905.07841}, note = {Source identifier: 1905.07841} }