@misc{indiciaeac79d8866d00, title = {CatLIP: CLIP-level Visual Recognition Accuracy with 2.7x Faster Pre-training on Web-scale Image-Text Data}, author = {Sachin Mehta and Maxwell Horton and Fartash Faghri and Mohammad Hossein Sekhavat and Mahyar Najibi and Mehrdad Farajtabar and Oncel Tuzel and Mohammad Rastegari}, year = {2024}, url = {https://arxiv.org/abs/2404.15653}, note = {Source identifier: 2404.15653} }