@misc{indiciaea416382c35ea, title = {ViT-CoMer: Vision Transformer with Convolutional Multi-scale Feature Interaction for Dense Predictions}, author = {Chunlong Xia and Xinliang Wang and Feng Lv and Xin Hao and Yifeng Shi}, year = {2024}, url = {https://arxiv.org/abs/2403.07392}, note = {Source identifier: 2403.07392} }