@misc{indiciae7f6e46b9f6b0, title = {Compound Tokens: Channel Fusion for Vision-Language Representation Learning}, author = {Maxwell Mbabilla Aladago and AJ Piergiovanni}, year = {2022}, url = {https://arxiv.org/abs/2212.01447}, note = {Source identifier: 2212.01447} }