@misc{indiciae60ebfb3ef3ea, title = {AFL-Net: Integrating Audio, Facial, and Lip Modalities with a Two-step Cross-attention for Robust Speaker Diarization in the Wild}, author = {Yongkang Yin and Xu Li and Ying Shan and Yuexian Zou}, year = {2024}, url = {https://arxiv.org/abs/2312.05730}, note = {Source identifier: 2312.05730} }