@misc{indiciae50039bae4757, title = {VidBot: Learning Generalizable 3D Actions from In-the-Wild 2D Human Videos for Zero-Shot Robotic Manipulation}, author = {Hanzhi Chen and Boyang Sun and Anran Zhang and Marc Pollefeys and Stefan Leutenegger}, year = {2025}, url = {https://arxiv.org/abs/2503.07135}, note = {Source identifier: 2503.07135} }