@misc{indiciaeb20e21135c1e, title = {Robust LLM-based Audio-Visual Speech Recognition with Sparse Modality Alignment and Visual Unit-Guided Refinement}, author = {Fei Su and Cancan Li and Juan Liu and Wei Ju and Hongbin Suo and Ming Li}, year = {2026}, url = {https://arxiv.org/abs/2603.03811}, note = {Source identifier: 2603.03811} }