@misc{indiciae41b93c0e46a4, title = {Text-Conditional JEPA for Learning Semantically Rich Visual Representations}, author = {Chen Huang and Xianhang Li and Vimal Thilak and Etai Littwin and Josh Susskind}, year = {2026}, url = {https://arxiv.org/abs/2605.03245}, note = {Source identifier: 2605.03245} }