@misc{indiciae4035b031f43c, title = {Multimodal Representation Loss Between Timed Text and Audio for Regularized Speech Separation}, author = {Tsun-An Hsieh and Heeyoul Choi and Minje Kim}, year = {2024}, url = {https://arxiv.org/abs/2406.08328}, note = {Source identifier: 2406.08328} }