@misc{indiciaebf1fa5ebd653, title = {DuoFormer: Leveraging Hierarchical Representations by Local and Global Attention Vision Transformer}, author = {Xiaoya Tang and Bodong Zhang and Man Minh Ho and Beatrice S. Knudsen and Tolga Tasdizen}, year = {2025}, url = {https://arxiv.org/abs/2506.12982}, note = {Source identifier: 2506.12982} }