@misc{indiciae39b1e11daf6b, title = {VLM Q-Learning: Aligning Vision-Language Models for Interactive Decision-Making}, author = {Jake Grigsby and Yuke Zhu and Michael Ryoo and Juan Carlos Niebles}, year = {2025}, url = {https://arxiv.org/abs/2505.03181}, note = {Source identifier: 2505.03181} }