@misc{indiciae5fae332c10c6, title = {Leveraging Gaze and Set-of-Mark in VLLMs for Human-Object Interaction Anticipation from Egocentric Videos}, author = {Daniele Materia and Francesco Ragusa and Giovanni Maria Farinella}, year = {2026}, url = {https://arxiv.org/abs/2604.03667}, note = {Source identifier: 2604.03667} }