@misc{indiciae44fce67920df, title = {TextHawk2: A Large Vision-Language Model Excels in Bilingual OCR and Grounding with 16x Fewer Tokens}, author = {Ya-Qi Yu and Minghui Liao and Jiwen Zhang and Jihao Wu}, year = {2024}, url = {https://arxiv.org/abs/2410.05261}, note = {Source identifier: 2410.05261} }