@misc{indiciaea91a5ba21b4e, title = {Precise Video-to-Audio Generation with Cross-Modal Alignment in Latent Space}, author = {Thanh V. T. Tran and Ngoc-Son Nguyen and Luong Tran and Long-Khanh Pham and Paarth Neekhara and Shehzeen Hussain and Van Nguyen}, year = {2026}, url = {https://arxiv.org/abs/2607.06405}, note = {Source identifier: 2607.06405} }