@misc{indiciae6843e3e5e4eb, title = {Audio-visual Controlled Video Diffusion with Masked Selective State Spaces Modeling for Natural Talking Head Generation}, author = {Fa-Ting Hong and Zunnan Xu and Zixiang Zhou and Jun Zhou and Xiu Li and Qin Lin and Qinglin Lu and Dan Xu}, year = {2025}, url = {https://arxiv.org/abs/2504.02542}, note = {Source identifier: 2504.02542} }