@article{tracktheanswerextendingtextvqafromimage, title = {Track the Answer: Extending TextVQA from Image to Video with Spatio-Temporal Clues}, author = {Yan Zhang and Gangyan Zeng and Huawen Shen and Daiqing Wu and Yu Zhou and Can Ma}, year = {2024}, eprint = {2412.12502}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2412.12502v1}, }