@misc{indiciae4195fecb2b6d, title = {X-DETR: A Versatile Architecture for Instance-wise Vision-Language Tasks}, author = {Zhaowei Cai and Gukyeong Kwon and Avinash Ravichandran and Erhan Bas and Zhuowen Tu and Rahul Bhotika and Stefano Soatto}, year = {2022}, url = {https://arxiv.org/abs/2204.05626}, note = {Source identifier: 2204.05626} }