@misc{indiciae2795f9088e90, title = {QLIP: Text-Aligned Visual Tokenization Unifies Auto-Regressive Multimodal Understanding and Generation}, author = {Yue Zhao and Fuzhao Xue and Scott Reed and Linxi Fan and Yuke Zhu and Jan Kautz and Zhiding Yu and Philipp Krähenbühl and De-An Huang}, year = {2025}, url = {https://arxiv.org/abs/2502.05178}, note = {Source identifier: 2502.05178} }