@article{3dcavlaleveragingdepthand3dcontextto, title = {3D CAVLA: Leveraging Depth and 3D Context to Generalize Vision Language Action Models for Unseen Tasks}, author = {Vineet Bhat and Yu-Hsiang Lan and Prashanth Krishnamurthy and Ramesh Karri and Farshad Khorrami}, year = {2025}, eprint = {2505.05800}, archivePrefix = {arXiv}, url = {https://arxiv.org/abs/2505.05800v1}, }