@misc{indiciae3a291d37c713, title = {DiVISe: Direct Visual-Input Speech Synthesis Preserving Speaker Characteristics And Intelligibility}, author = {Yifan Liu and Yu Fang and Zhouhan Lin}, year = {2025}, url = {https://arxiv.org/abs/2503.05223}, note = {Source identifier: 2503.05223} }