@misc{indiciae6315bf2f47d5, title = {OG-VLA: Orthographic Image Generation for 3D-Aware Vision-Language Action Model}, author = {Ishika Singh and Ankit Goyal and Stan Birchfield and Dieter Fox and Animesh Garg and Valts Blukis}, year = {2025}, url = {https://arxiv.org/abs/2506.01196}, note = {Source identifier: 2506.01196} }