@misc{indiciaeded7325a88f0, title = {Towards Expressive and Faithful Audio-to-Image Generation: A Unified Multimodal Dataset and Synthesis Framework}, author = {Dongxu Ge and Shansong Liu and Cheng Gong and Xiao-Lei Zhang and Chi Zhang and Xuelong Li}, year = {2026}, doi = {10.1145/3767308.3836313}, url = {https://arxiv.org/abs/2608.09529}, note = {Source identifier: 2608.09529} }