@misc{indiciae277d8b9c270f, title = {STNet: Deep Audio-Visual Fusion Network for Robust Speaker Tracking}, author = {Yidi Li and Hong Liu and Bing Yang}, year = {2024}, url = {https://arxiv.org/abs/2410.05964}, note = {Source identifier: 2410.05964} }