@misc{indiciae8a136d88b7c8, title = {Improving Audio Captioning Models with Fine-grained Audio Features, Text Embedding Supervision, and LLM Mix-up Augmentation}, author = {Shih-Lun Wu and Xuankai Chang and Gordon Wichern and Jee-weon Jung and François Germain and Jonathan Le Roux and Shinji Watanabe}, year = {2024}, url = {https://arxiv.org/abs/2309.17352}, note = {Source identifier: 2309.17352} }