@misc{indiciae1f75181f800d, title = {AudioSetCaps: An Enriched Audio-Caption Dataset using Automated Generation Pipeline with Large Audio and Language Models}, author = {Jisheng Bai and Haohe Liu and Mou Wang and Dongyuan Shi and Wenwu Wang and Mark D. Plumbley and Woon-Seng Gan and Jianfeng Chen}, year = {2024}, url = {https://arxiv.org/abs/2411.18953}, note = {Source identifier: 2411.18953} }