@misc{indiciaebf5e575d5408, title = {N3D-VLM: Native 3D Grounding Enables Accurate Spatial Reasoning in Vision-Language Models}, author = {Yuxin Wang and Lei Ke and Boqiang Zhang and Tianyuan Qu and Hanxun Yu and Zhenpeng Huang and Meng Yu and Dan Xu and Dong Yu}, year = {2025}, url = {https://arxiv.org/abs/2512.16561}, note = {Source identifier: 2512.16561} }