@misc{indiciae8f697c3a6ee8, title = {Optimizing Multimodal LLMs for Egocentric Video Understanding: A Solution for the HD-EPIC VQA Challenge}, author = {Sicheng Yang and Yukai Huang and Shitong Sun and Weitong Cai and Jiankang Deng and Jifei Song and Zhensong Zhang}, year = {2026}, url = {https://arxiv.org/abs/2601.10228}, note = {Source identifier: 2601.10228} }