@misc{indiciaea36437ba832d, title = {Multi-modal and Multi-scale Spatial Environment Understanding for Immersive Visual Text-to-Speech}, author = {Rui Liu and Shuwei He and Yifan Hu and Haizhou Li}, year = {2025}, url = {https://arxiv.org/abs/2412.11409}, note = {Source identifier: 2412.11409} }