@misc{indiciaedcae613bbc17, title = {DeepAudio-V1:Towards Multi-Modal Multi-Stage End-to-End Video to Speech and Audio Generation}, author = {Haomin Zhang and Chang Liu and Junjie Zheng and Zihao Chen and Chaofan Ding and Xinhan Di}, year = {2025}, url = {https://arxiv.org/abs/2503.22265}, note = {Source identifier: 2503.22265} }