@misc{indiciae2e133399c9a1, title = {Enhance Multimodal Consistency and Coherence for Text-Image Plan Generation}, author = {Xiaoxin Lu and Ranran Haoran Zhang and Yusen Zhang and Rui Zhang}, year = {2025}, url = {https://arxiv.org/abs/2506.11380}, note = {Source identifier: 2506.11380} }