@misc{indiciaed4033722a37f, title = {Beyond Lips: Integrating Gesture and Lip Cues for Robust Audio-visual Speaker Extraction}, author = {Zexu Pan and Xinyuan Qian and Shengkui Zhao and Kun Zhou and Bin Ma}, year = {2026}, url = {https://arxiv.org/abs/2601.19130}, note = {Source identifier: 2601.19130} }