@misc{indiciaec36ceb855888, title = {Mechanisms of Multimodal Synchronization: Insights from Decoder-Based Video-Text-to-Speech Synthesis}, author = {Akshita Gupta and Tatiana Likhomanenko and Karren Dai Yang and Richard He Bai and Zakaria Aldeneh and Navdeep Jaitly}, year = {2026}, url = {https://arxiv.org/abs/2411.17690}, note = {Source identifier: 2411.17690} }