@misc{indiciae45e84b199f05, title = {AutoMoT: A Unified Vision-Language-Action Model with Asynchronous Mixture-of-Transformers for End-to-End Autonomous Driving}, author = {Wenhui Huang and Songyan Zhang and Qihang Huang and Zhidong Wang and Zhiqi Mao and Collister Chua and Zhan Chen and Long Chen and Chen Lv}, year = {2026}, url = {https://arxiv.org/abs/2603.14851}, note = {Source identifier: 2603.14851} }