@misc{indiciae3be4dc21a8ee, title = {Phoneme-Level Visual Speech Recognition via Point-Visual Fusion and Language Model Reconstruction}, author = {Matthew Kit Khinn Teng and Haibo Zhang and Takeshi Saitoh}, year = {2026}, url = {https://arxiv.org/abs/2507.18863}, note = {Source identifier: 2507.18863} }