@misc{indiciaed111508c8ff8, title = {Sound to Visual Scene Generation by Audio-to-Visual Latent Alignment}, author = {Kim Sung-Bin and Arda Senocak and Hyunwoo Ha and Andrew Owens and Tae-Hyun Oh}, year = {2023}, url = {https://arxiv.org/abs/2303.17490}, note = {Source identifier: 2303.17490} }