@misc{indiciaed89b3f443918, title = {DiffV2S: Diffusion-based Video-to-Speech Synthesis with Vision-guided Speaker Embedding}, author = {Jeongsoo Choi and Joanna Hong and Yong Man Ro}, year = {2023}, url = {https://arxiv.org/abs/2308.07787}, note = {Source identifier: 2308.07787} }