@misc{indiciae988611538d53, title = {VisualSpeech: Enhancing Prosody Modeling in TTS Using Video}, author = {Shumin Que and Anton Ragni}, year = {2025}, url = {https://arxiv.org/abs/2501.19258}, note = {Source identifier: 2501.19258} }