@misc{indiciae4aad4f6fd294, title = {Learning Joint Representations of Videos and Sentences with Web Image Search}, author = {Mayu Otani and Yuta Nakashima and Esa Rahtu and Janne Heikkilä and Naokazu Yokoya}, year = {2016}, url = {https://arxiv.org/abs/1608.02367}, note = {Source identifier: 1608.02367} }