@misc{indiciaec942cc4c5221, title = {PixCLIP: Achieving Fine-grained Visual Language Understanding via Any-granularity Pixel-Text Alignment Learning}, author = {Yicheng Xiao and Yu Chen and Haoxuan Ma and Jiale Hong and Caorui Li and Lingxiang Wu and Haiyun Guo and Jinqiao Wang}, year = {2025}, url = {https://arxiv.org/abs/2511.04601}, note = {Source identifier: 2511.04601} }