@misc{indiciaebda1d7d4da8e, title = {UniTAF: A Modular Framework for Joint Text-to-Speech and Audio-to-Face Modeling}, author = {Qiangong Zhou and Nagasaka Tomohiro}, year = {2026}, url = {https://arxiv.org/abs/2602.15651}, note = {Source identifier: 2602.15651} }