@misc{indiciaebe5667b74f1a, title = {ImmersiveTTS: Environment-Aware Text-to-Speech with Multimodal Diffusion Transformer and Domain-Specific Representation Alignment}, author = {Jun-Hak Yun and Seung-Bin Kim and Seong-Whan Lee}, year = {2026}, url = {https://arxiv.org/abs/2605.30965}, note = {Source identifier: 2605.30965} }