@misc{indiciaed73ae3c1ca42, title = {Attention or Convolution: Transformer Encoders in Audio Language Models for Inference Efficiency}, author = {Sungho Jeon and Ching-Feng Yeh and Hakan Inan and Wei-Ning Hsu and Rashi Rungta and Yashar Mehdad and Daniel Bikel}, year = {2024}, url = {https://arxiv.org/abs/2311.02772}, note = {Source identifier: 2311.02772} }