@misc{indiciae626eef295ae5, title = {Improving Medical Speech-to-Text Accuracy with Vision-Language Pre-training Model}, author = {Jaeyoung Huh and Sangjoon Park and Jeong Eun Lee and Jong Chul Ye}, year = {2023}, url = {https://arxiv.org/abs/2303.00091}, note = {Source identifier: 2303.00091} }