@misc{indiciae7ab402ebfb3d, title = {Enhancing Temporal Understanding in Video-LLMs through Stacked Temporal Attention in Vision Encoders}, author = {Ali Rasekh and Erfan Bagheri Soula and Omid Daliran and Simon Gottschalk and Mohsen Fayyaz}, year = {2025}, url = {https://arxiv.org/abs/2510.26027}, note = {Source identifier: 2510.26027} }