@misc{indiciae97410825d70f, title = {An Approach to Combining Video and Speech with Large Language Models in Human-Robot Interaction}, author = {Guanting Shen and Zi Tian}, year = {2026}, url = {https://arxiv.org/abs/2602.20219}, note = {Source identifier: 2602.20219} }