@misc{indiciae7a1bb9bbf1f9, title = {LLaVAction: evaluating and training multi-modal large language models for action understanding}, author = {Haozhe Qi and Shaokai Ye and Alexander Mathis and Mackenzie W. Mathis}, year = {2026}, url = {https://arxiv.org/abs/2503.18712}, note = {Source identifier: 2503.18712} }