@misc{indiciaeeb0663c9a8dd, title = {Tell What You Hear From What You See -- Video to Audio Generation Through Text}, author = {Xiulong Liu and Kun Su and Eli Shlizerman}, year = {2025}, url = {https://arxiv.org/abs/2411.05679}, note = {Source identifier: 2411.05679} }