@misc{indiciae72957b66dce0, title = {LLaVA-Mini: Efficient Image and Video Large Multimodal Models with One Vision Token}, author = {Shaolei Zhang and Qingkai Fang and Zhe Yang and Yang Feng}, year = {2025}, url = {https://arxiv.org/abs/2501.03895}, note = {Source identifier: 2501.03895} }