@misc{indiciaec21bf9d5e957, title = {Inf-MLLM: Efficient Streaming Inference of Multimodal Large Language Models on a Single GPU}, author = {Zhenyu Ning and Jieru Zhao and Qihao Jin and Wenchao Ding and Minyi Guo}, year = {2024}, url = {https://arxiv.org/abs/2409.09086}, note = {Source identifier: 2409.09086} }