@misc{indiciae7c45022ed20e, title = {V-CASS: Vision-context-aware Expressive Speech Synthesis for Enhancing User Understanding of Videos}, author = {Qixin Wang and Songtao Zhou and Zeyu Jin and Chenglin Guo and Shikun Sun and Xiaoyu Qin}, year = {2025}, url = {https://arxiv.org/abs/2506.16716}, note = {Source identifier: 2506.16716} }