@misc{indiciae6e7796f99e61, title = {From Semantics to Pixels: Coarse-to-Fine Masked Autoencoders for Hierarchical Visual Understanding}, author = {Wenzhao Xiang and Yue Wu and Hongyang Yu and Feng Gao and Fan Yang and Xilin Chen}, year = {2026}, url = {https://arxiv.org/abs/2603.09955}, note = {Source identifier: 2603.09955} }