@misc{indiciae935e74194d62, title = {PixFoundation 2.0: Do Video Multi-Modal LLMs Use Motion in Visual Grounding?}, author = {Mennatullah Siam}, year = {2025}, url = {https://arxiv.org/abs/2509.02807}, note = {Source identifier: 2509.02807} }