@misc{indiciaed90af04ccb4a, title = {Action QFormer: Structured Representation Shaping under Action Supervision in Vision-Language-Action Models}, author = {Yufeng Ji and Wenhao Tang and Haoyi Niu and Koushil Sreenath and Yi Wu and Zhongyu Li}, year = {2026}, url = {https://arxiv.org/abs/2607.14635}, note = {Source identifier: 2607.14635} }