@misc{indiciae2a2951522c73, title = {Stream-Omni: Simultaneous Multimodal Interactions with Large Language-Vision-Speech Model}, author = {Shaolei Zhang and Shoutao Guo and Qingkai Fang and Yan Zhou and Yang Feng}, year = {2025}, url = {https://arxiv.org/abs/2506.13642}, note = {Source identifier: 2506.13642} }