@misc{indiciaefdc25587073c, title = {ST-VLM: Kinematic Instruction Tuning for Spatio-Temporal Reasoning in Vision-Language Models}, author = {Dohwan Ko and Sihyeon Kim and Yumin Suh and Vijay Kumar B. G and Minseo Yoon and Manmohan Chandraker and Hyunwoo J. Kim}, year = {2025}, url = {https://arxiv.org/abs/2503.19355}, note = {Source identifier: 2503.19355} }