@misc{indiciae1b8032c88a4b, title = {Q: How to Specialize Large Vision-Language Models to Data-Scarce VQA Tasks? A: Self-Train on Unlabeled Images!}, author = {Zaid Khan and Vijay Kumar BG and Samuel Schulter and Xiang Yu and Yun Fu and Manmohan Chandraker}, year = {2023}, url = {https://arxiv.org/abs/2306.03932}, note = {Source identifier: 2306.03932} }