@misc{indiciae884f90e36a49, title = {Unifying Vision-Language Latents for Zero-label Image Caption Enhancement}, author = {Sanghyun Byun and Jung Ick Guack and Mohanad Odema and Baisub Lee and Jacob Song and Woo Seong Chung}, year = {2025}, url = {https://arxiv.org/abs/2510.12931}, note = {Source identifier: 2510.12931} }