@misc{indiciae982d805177b0, title = {Modulating CNN Features with Pre-Trained ViT Representations for Open-Vocabulary Object Detection}, author = {Xiangyu Gao and Yu Dai and Benliu Qiu and Lanxiao Wang and Heqian Qiu and Hongliang Li}, year = {2025}, url = {https://arxiv.org/abs/2501.16981}, note = {Source identifier: 2501.16981} }