@misc{indiciae311e64bfbba7, title = {SVLTA: Benchmarking Vision-Language Temporal Alignment via Synthetic Video Situation}, author = {Hao Du and Bo Wu and Yan Lu and Zhendong Mao}, year = {2025}, url = {https://arxiv.org/abs/2504.05925}, note = {Source identifier: 2504.05925} }