@misc{indiciae07a0fc7acd14, title = {Track the Answer: Extending TextVQA from Image to Video with Spatio-Temporal Clues}, author = {Yan Zhang and Gangyan Zeng and Huawen Shen and Daiqing Wu and Yu Zhou and Can Ma}, year = {2024}, url = {https://arxiv.org/abs/2412.12502}, note = {Source identifier: 2412.12502} }