@misc{indiciae3060ef8cdbb8, title = {Large Language Models are Strong Audio-Visual Speech Recognition Learners}, author = {Umberto Cappellazzo and Minsu Kim and Honglie Chen and Pingchuan Ma and Stavros Petridis and Daniele Falavigna and Alessio Brutti and Maja Pantic}, year = {2025}, url = {https://arxiv.org/abs/2409.12319}, note = {Source identifier: 2409.12319} }