@misc{indiciae8dac02c59655, title = {Speech-Omni-Lite: Portable Speech Interfaces for Vision-Language Models}, author = {Dehua Tao and Xuan Luo and Daxin Tan and Kai Chen and Lanqing Hong and Jing Li and Ruifeng Xu and Xiao Chen}, year = {2026}, url = {https://arxiv.org/abs/2603.09627}, note = {Source identifier: 2603.09627} }