@misc{indiciae976040206f6c, title = {HandsOnVLM: Vision-Language Models for Hand-Object Interaction Prediction}, author = {Chen Bao and Jiarui Xu and Xiaolong Wang and Abhinav Gupta and Homanga Bharadhwaj}, year = {2024}, url = {https://arxiv.org/abs/2412.13187}, note = {Source identifier: 2412.13187} }