@misc{indiciaeee9521f547ce, title = {DiffAVA: Personalized Text-to-Audio Generation with Visual Alignment}, author = {Shentong Mo and Jing Shi and Yapeng Tian}, year = {2023}, url = {https://arxiv.org/abs/2305.12903}, note = {Source identifier: 2305.12903} }