@misc{indiciae51344cbfa804, title = {Learning to Generate Grounded Visual Captions without Localization Supervision}, author = {Chih-Yao Ma and Yannis Kalantidis and Ghassan AlRegib and Peter Vajda and Marcus Rohrbach and Zsolt Kira}, year = {2020}, url = {https://arxiv.org/abs/1906.00283}, note = {Source identifier: 1906.00283} }