@misc{indiciae11fb80ba3976, title = {Embodied Image Captioning: Self-supervised Learning Agents for Spatially Coherent Image Descriptions}, author = {Tommaso Galliena and Tommaso Apicella and Stefano Rosa and Pietro Morerio and Alessio Del Bue and Lorenzo Natale}, year = {2025}, url = {https://arxiv.org/abs/2504.08531}, note = {Source identifier: 2504.08531} }