@misc{indiciae450e1e6f5607, title = {Text as Images: Can Multimodal Large Language Models Follow Printed Instructions in Pixels?}, author = {Xiujun Li and Yujie Lu and Zhe Gan and Jianfeng Gao and William Yang Wang and Yejin Choi}, year = {2024}, url = {https://arxiv.org/abs/2311.17647}, note = {Source identifier: 2311.17647} }