@misc{indiciae3e542df0c42d, title = {ViMoNet: A Multimodal Vision-Language Framework for Human Behavior Understanding from Motion and Video}, author = {Rajan Das Gupta and Lei Wei and Md Yeasin Rahat and Nafiz Fahad and Abir Ahmed and Liew Tze Hui}, year = {2026}, url = {https://arxiv.org/abs/2508.09818}, note = {Source identifier: 2508.09818} }