@misc{indiciaef7611f26d78f, title = {Audio-visual speech separation based on joint feature representation with cross-modal attention}, author = {Junwen Xiong and Peng Zhang and Lei Xie and Wei Huang and Yufei Zha and Yanning Zhang}, year = {2022}, url = {https://arxiv.org/abs/2203.02655}, note = {Source identifier: 2203.02655} }