@misc{indiciae2520266c71df, title = {Proxy3D: Efficient 3D Representations for Vision-Language Models via Semantic Clustering and Alignment}, author = {Jerry Jiang and Haowen Sun and Denis Gudovskiy and Yohei Nakata and Tomoyuki Okuno and Kurt Keutzer and Wenzhao Zheng}, year = {2026}, url = {https://arxiv.org/abs/2605.08064}, note = {Source identifier: 2605.08064} }