@misc{indiciae0396b6bfc145, title = {VAPO: End-to-end Slide-Enhanced Speech Recognition with Omni-modal Large Language Models}, author = {Rui Hu and Delai Qiu and Yining Wang and Shengping Liu and Jitao Sang}, year = {2026}, url = {https://arxiv.org/abs/2510.08618}, note = {Source identifier: 2510.08618} }