@misc{indiciaea2499ae44504, title = {Deep Visual Forced Alignment: Learning to Align Transcription with Talking Face Video}, author = {Minsu Kim and Chae Won Kim and Yong Man Ro}, year = {2023}, url = {https://arxiv.org/abs/2303.08670}, note = {Source identifier: 2303.08670} }