@misc{indiciae5025c2c8a366, title = {Speak While Watching: Unleashing TRUE Real-Time Video Understanding Capability of Multimodal Large Language Models}, author = {Junyan Lin and Junlong Tong and Hao Wu and Jialiang Zhang and Jinming Liu and Xin Jin and Xiaoyu Shen}, year = {2026}, url = {https://arxiv.org/abs/2601.06843}, note = {Source identifier: 2601.06843} }