@misc{indiciaed384707caaab, title = {Light Future: Multimodal Action Frame Prediction via InstructPix2Pix}, author = {Zesen Zhong and Duomin Zhang and Yijia Li}, year = {2025}, url = {https://arxiv.org/abs/2507.14809}, note = {Source identifier: 2507.14809} }