@misc{indiciaef66819b5cfc4, title = {CineSRD: Leveraging Visual, Acoustic, and Linguistic Cues for Open-World Visual Media Speaker Diarization}, author = {Liangbin Huang and Xiaohua Liao and Chaoqun Cui and Shijing Wang and Zhaolong Huang and Yanlong Du and Wenji Mao}, year = {2026}, url = {https://arxiv.org/abs/2603.16966}, note = {Source identifier: 2603.16966} }