@misc{indiciae3e55ec6db7ff, title = {ROVER: Recursive Reasoning Over Videos with Vision-Language Models for Embodied Tasks}, author = {Philip Schroeder and Ondrej Biza and Thomas Weng and Hongyin Luo and James Glass}, year = {2025}, url = {https://arxiv.org/abs/2508.01943}, note = {Source identifier: 2508.01943} }