@misc{indiciae6a0f7e548586, title = {Facetron: A Multi-speaker Face-to-Speech Model based on Cross-modal Latent Representations}, author = {Se-Yun Um and Jihyun Kim and Jihyun Lee and Hong-Goo Kang}, year = {2023}, url = {https://arxiv.org/abs/2107.12003}, note = {Source identifier: 2107.12003} }