@misc{indiciae803ddc2de3e9, title = {OTTER: A Vision-Language-Action Model with Text-Aware Visual Feature Extraction}, author = {Huang Huang and Fangchen Liu and Letian Fu and Tingfan Wu and Mustafa Mukadam and Jitendra Malik and Ken Goldberg and Pieter Abbeel}, year = {2025}, url = {https://arxiv.org/abs/2503.03734}, note = {Source identifier: 2503.03734} }