@misc{indiciae2b8aa9036978, title = {V2PE: Improving Multimodal Long-Context Capability of Vision-Language Models with Variable Visual Position Encoding}, author = {Junqi Ge and Ziyi Chen and Jintao Lin and Jinguo Zhu and Xihui Liu and Jifeng Dai and Xizhou Zhu}, year = {2024}, url = {https://arxiv.org/abs/2412.09616}, note = {Source identifier: 2412.09616} }