@misc{indiciaeb88893530f06, title = {VT-LVLM-AR: A Video-Temporal Large Vision-Language Model Adapter for Fine-Grained Action Recognition in Long-Term Videos}, author = {Kaining Li and Shuwei He and Zihan Xu}, year = {2025}, url = {https://arxiv.org/abs/2508.15903}, note = {Source identifier: 2508.15903} }