@misc{indiciae54cc80fa2dd9, title = {Spatially-Grounded Text-to-Video Generation via Inference-Time Gradient-Free Optimization}, author = {Guillaume Jeanneret and Mathis Koroglu and Hugo Caselles-Dupré and Arnaud Dapogny and Matthieu Cord}, year = {2026}, url = {https://arxiv.org/abs/2608.13037}, note = {Source identifier: 2608.13037} }