@misc{indiciaee33c78c84905, title = {Show and Speak: Directly Synthesize Spoken Description of Images}, author = {Xinsheng Wang and Siyuan Feng and Jihua Zhu and Mark Hasegawa-Johnson and Odette Scharenborg}, year = {2020}, url = {https://arxiv.org/abs/2010.12267}, note = {Source identifier: 2010.12267} }