@misc{indiciae8a91e3579506, title = {Joint Training or Not: An Exploration of Pre-trained Speech Models in Audio-Visual Speaker Diarization}, author = {Huan Zhao and Li Zhang and Yue Li and Yannan Wang and Hongji Wang and Wei Rao and Qing Wang and Lei Xie}, year = {2023}, url = {https://arxiv.org/abs/2312.04131}, note = {Source identifier: 2312.04131} }