@misc{indiciaed72c868c24bc, title = {SPEECH-COCO: 600k Visually Grounded Spoken Captions Aligned to MSCOCO Data Set}, author = {William Havard and Laurent Besacier and Olivier Rosec}, year = {2020}, doi = {10.21437/glu.2017-9}, url = {https://arxiv.org/abs/1707.08435}, note = {Source identifier: 1707.08435} }