@article{automotaunifiedvisionlanguageactionmodel, title = {AutoMoT: A Unified Vision-Language-Action Model with Asynchronous Mixture-of-Transformers for End-to-End Autonomous Driving}, author = {Wenhui Huang and Songyan Zhang and Qihang Huang and Zhidong Wang and Zhiqi Mao and Collister Chua and Zhan Chen and Long Chen and Chen Lv}, year = {2026}, eprint = {2603.14851}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2603.14851}, }