@misc{indiciae834fa18eb88c, title = {Pixel-level Scene Understanding in One Token: Visual States Need What-is-Where Composition}, author = {Seokmin Lee and Yunghee Lee and Byeonghyun Pak and Byeongju Woo}, year = {2026}, url = {https://arxiv.org/abs/2603.13904}, note = {Source identifier: 2603.13904} }