@misc{indiciaeb2feeec24aa5, title = {CueNet: Robust Audio-Visual Speaker Extraction through Cross-Modal Cue Mining and Interaction}, author = {Jiadong Wang and Ke Zhang and Xinyuan Qian and Ruijie Tao and Haizhou Li and Björn Schuller}, year = {2026}, url = {https://arxiv.org/abs/2603.01530}, note = {Source identifier: 2603.01530} }