@misc{indiciae2a7bb65223f6, title = {Plug-and-Play Co-Occurring Face Attention for Robust Audio-Visual Speaker Extraction}, author = {Zexu Pan and Shengkui Zhao and Tingting Wang and Kun Zhou and Yukun Ma and Chong Zhang and Bin Ma}, year = {2025}, url = {https://arxiv.org/abs/2505.20635}, note = {Source identifier: 2505.20635} }