@misc{indiciae6781e045b388, title = {VideoLoom: A Video Large Language Model for Joint Spatial-Temporal Understanding}, author = {Jiapeng Shi and Junke Wang and Zuyao You and Bo He and Zuxuan Wu}, year = {2026}, url = {https://arxiv.org/abs/2601.07290}, note = {Source identifier: 2601.07290} }