@misc{indiciae4b0884232eea, title = {Separate in the Speech Chain: Cross-Modal Conditional Audio-Visual Target Speech Extraction}, author = {Zhaoxi Mu and Xinyu Yang}, year = {2024}, url = {https://arxiv.org/abs/2404.12725}, note = {Source identifier: 2404.12725} }