@misc{indiciaef7cb79c23d50, title = {Attention-based cross-modal fusion for audio-visual voice activity detection in musical video streams}, author = {Yuanbo Hou and Zhesong Yu and Xia Liang and Xingjian Du and Bilei Zhu and Zejun Ma and Dick Botteldooren}, year = {2021}, url = {https://arxiv.org/abs/2106.11411}, note = {Source identifier: 2106.11411} }