@misc{indiciae14e660774cf8, title = {Stateful Visual Encoders for Vision-Language Models}, author = {Zirui Wang and Junwei Yu and Adam Yala and David M. Chan and Joseph E. Gonzalez and Trevor Darrell}, year = {2026}, url = {https://arxiv.org/abs/2606.04433}, note = {Source identifier: 2606.04433} }