@misc{indiciae6d876a894bc4, title = {Real-Time Audio-Visual Speech Enhancement Using Pre-trained Visual Representations}, author = {T. Aleksandra Ma and Sile Yin and Li-Chia Yang and Shuo Zhang}, year = {2025}, url = {https://arxiv.org/abs/2507.21448}, note = {Source identifier: 2507.21448} }