@misc{indiciae715d404e0b1c, title = {From Vision to Audio and Beyond: A Unified Model for Audio-Visual Representation and Generation}, author = {Kun Su and Xiulong Liu and Eli Shlizerman}, year = {2024}, url = {https://arxiv.org/abs/2409.19132}, note = {Source identifier: 2409.19132} }