@misc{indiciae0824f63ba114, title = {From Faces to Voices: Learning Hierarchical Representations for High-quality Video-to-Speech}, author = {Ji-Hoon Kim and Jeongsoo Choi and Jaehun Kim and Chaeyoung Jung and Joon Son Chung}, year = {2025}, url = {https://arxiv.org/abs/2503.16956}, note = {Source identifier: 2503.16956} }