@misc{indiciae4abfef3750eb, title = {Integrating Self-supervised Speech Model with Pseudo Word-level Targets from Visually-grounded Speech Model}, author = {Hung-Chieh Fang and Nai-Xuan Ye and Yi-Jen Shih and Puyuan Peng and Hsuan-Fu Wang and Layne Berry and Hung-yi Lee and David Harwath}, year = {2024}, url = {https://arxiv.org/abs/2402.05819}, note = {Source identifier: 2402.05819} }