@misc{indiciaead4e270bd640, title = {Inverse-LLaVA: Rethinking Multimodal Alignment via Text-to-Vision Mapping}, author = {Xuhui Zhan and Tyler Derr}, year = {2026}, url = {https://arxiv.org/abs/2508.12466}, note = {Source identifier: 2508.12466} }