@misc{indiciae5e48cce5b29d, title = {Seeing What You Say: Expressive Image Generation from Speech}, author = {Jiyoung Lee and Song Park and Sanghyuk Chun and Soo-Whan Chung}, year = {2025}, url = {https://arxiv.org/abs/2511.03423}, note = {Source identifier: 2511.03423} }