@misc{indiciaef0aedb976305, title = {MiniDrive: More Efficient Vision-Language Models with Multi-Level 2D Features as Text Tokens for Autonomous Driving}, author = {Enming Zhang and Xingyuan Dai and Min Huang and Yisheng Lv and Qinghai Miao}, year = {2025}, url = {https://arxiv.org/abs/2409.07267}, note = {Source identifier: 2409.07267} }