@misc{indiciaed64c3fe94c7d, title = {GroundVLP: Harnessing Zero-shot Visual Grounding from Vision-Language Pre-training and Open-Vocabulary Object Detection}, author = {Haozhan Shen and Tiancheng Zhao and Mingwei Zhu and Jianwei Yin}, year = {2023}, url = {https://arxiv.org/abs/2312.15043}, note = {Source identifier: 2312.15043} }