@misc{indiciae4e3eb5382214, title = {From Images to Words: Efficient Cross-Modal Knowledge Distillation to Language Models from Black-box Teachers}, author = {Ayan Sengupta and Shantanu Dixit and Md Shad Akhtar and Tanmoy Chakraborty}, year = {2026}, url = {https://arxiv.org/abs/2603.10877}, note = {Source identifier: 2603.10877} }