@article{stvlmkinematicinstructiontuningfor, title = {ST-VLM: Kinematic Instruction Tuning for Spatio-Temporal Reasoning in Vision-Language Models}, author = {Dohwan Ko and Sihyeon Kim and Yumin Suh and Vijay Kumar B. G and Minseo Yoon and Manmohan Chandraker and Hyunwoo J. Kim}, year = {2025}, eprint = {2503.19355}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2503.19355v2}, }