@misc{indiciae284886b7643b, title = {Semantic visually-guided acoustic highlighting with large vision-language models}, author = {Junhua Huang and Chao Huang and Chenliang Xu}, year = {2026}, url = {https://arxiv.org/abs/2601.08871}, note = {Source identifier: 2601.08871} }