@misc{indiciae594f0a5e3e09, title = {3D CAVLA: Leveraging Depth and 3D Context to Generalize Vision Language Action Models for Unseen Tasks}, author = {Vineet Bhat and Yu-Hsiang Lan and Prashanth Krishnamurthy and Ramesh Karri and Farshad Khorrami}, year = {2026}, url = {https://arxiv.org/abs/2505.05800}, note = {Source identifier: 2505.05800} }