@misc{indiciae5b5a5bcadd4c, title = {ReasonCLIP-58M: Visually Grounded Commonsense Reasoning Supervision for CLIP}, author = {Sicheng Zhang and Muzammal Naseer and Binzhu Xie and Naufal Suryanto and Shi Qiu and Jamal Bentahar and Naveed Akhtar and Mubarak Shah}, year = {2026}, url = {https://arxiv.org/abs/2606.26794}, note = {Source identifier: 2606.26794} }