@misc{indiciaef845b4310e7e, title = {Vision-Language Models for Egocentric Video: From Hand-Object Interaction to Embodied AI}, author = {Mohammad Zamani and Fatemeh Ziaeetabar}, year = {2026}, url = {https://arxiv.org/abs/2608.18671}, note = {Source identifier: 2608.18671} }