@misc{indiciae24f62bb9ad4c, title = {TokenFocus-VQA: Enhancing Text-to-Image Alignment with Position-Aware Focus and Multi-Perspective Aggregations on LVLMs}, author = {Zijian Zhang and Xuhui Zheng and Xuecheng Wu and Chong Peng and Xuezhi Cao}, year = {2025}, url = {https://arxiv.org/abs/2504.07556}, note = {Source identifier: 2504.07556} }