@misc{indiciae1b1412e05df0, title = {Transcription-Enriched Joint Embeddings for Spoken Descriptions of Images and Videos}, author = {Benet Oriol and Jordi Luque and Ferran Diego and Xavier Giro-i-Nieto}, year = {2020}, url = {https://arxiv.org/abs/2006.00785}, note = {Source identifier: 2006.00785} }