@misc{indiciae36493dce4cb0, title = {JARVIS-VLA: Post-Training Large-Scale Vision Language Models to Play Visual Games with Keyboards and Mouse}, author = {Muyao Li and Zihao Wang and Kaichen He and Xiaojian Ma and Yitao Liang}, year = {2025}, url = {https://arxiv.org/abs/2503.16365}, note = {Source identifier: 2503.16365} }