@misc{indiciaec7a06ef153aa, title = {VLM See, Robot Do: Human Demo Video to Robot Action Plan via Vision Language Model}, author = {Beichen Wang and Juexiao Zhang and Shuwen Dong and Irving Fang and Chen Feng}, year = {2025}, url = {https://arxiv.org/abs/2410.08792}, note = {Source identifier: 2410.08792} }