@misc{indiciae0fb3c648c42d, title = {EmbodiedMidtrain: Bridging the Gap between Vision-Language Models and Vision-Language-Action Models via Mid-training}, author = {Yiyang Du and Zhanqiu Guo and Xin Ye and Liu Ren and Chenyan Xiong}, year = {2026}, url = {https://arxiv.org/abs/2604.20012}, note = {Source identifier: 2604.20012} }