@misc{indiciaee7efeda55e64, title = {ResNetVLLM -- Multi-modal Vision LLM for the Video Understanding Task}, author = {Ahmad Khalil and Mahmoud Khalil and Alioune Ngom}, year = {2025}, url = {https://arxiv.org/abs/2504.14432}, note = {Source identifier: 2504.14432} }