@misc{indiciae1f7246c3bf8b, title = {Text-Free Image-to-Speech Synthesis Using Learned Segmental Units}, author = {Wei-Ning Hsu and David Harwath and Christopher Song and James Glass}, year = {2020}, url = {https://arxiv.org/abs/2012.15454}, note = {Source identifier: 2012.15454} }