@misc{indiciae41dd6b25ba7c, title = {MMS-LLaMA: Efficient LLM-based Audio-Visual Speech Recognition with Minimal Multimodal Speech Tokens}, author = {Jeong Hun Yeo and Hyeongseop Rha and Se Jin Park and Yong Man Ro}, year = {2025}, url = {https://arxiv.org/abs/2503.11315}, note = {Source identifier: 2503.11315} }