@misc{indiciae6f2954effae4, title = {Semantically consistent Video-to-Audio Generation using Multimodal Language Large Model}, author = {Gehui Chen and Guan'an Wang and Xiaowen Huang and Jitao Sang}, year = {2024}, url = {https://arxiv.org/abs/2404.16305}, note = {Source identifier: 2404.16305} }