@misc{indiciae82302c30c6c4, title = {Sound2Vision: Generating Diverse Visuals from Audio through Cross-Modal Latent Alignment}, author = {Kim Sung-Bin and Arda Senocak and Hyunwoo Ha and Tae-Hyun Oh}, year = {2024}, url = {https://arxiv.org/abs/2412.06209}, note = {Source identifier: 2412.06209} }