@misc{indiciae992dff588df1, title = {FOCUS: Internal MLLM Representations for Efficient Fine-Grained Visual Question Answering}, author = {Liangyu Zhong and Fabio Rosenthal and Joachim Sicking and Fabian Hüger and Thorsten Bagdonat and Hanno Gottschalk and Leo Schwinn}, year = {2025}, url = {https://arxiv.org/abs/2506.21710}, note = {Source identifier: 2506.21710} }