@misc{indiciae24e6c8c55540, title = {Pix2Cap-COCO: Advancing Visual Comprehension via Pixel-Level Captioning}, author = {Zuyao You and Junke Wang and Lingyu Kong and Bo He and Zuxuan Wu}, year = {2025}, url = {https://arxiv.org/abs/2501.13893}, note = {Source identifier: 2501.13893} }