@misc{indiciae48830c293296, title = {Audio-text Retrieval with Transformer-based Hierarchical Alignment and Disentangled Cross-modal Representation}, author = {Yifei Xin and Zhihong Zhu and Xuxin Cheng and Xusheng Yang and Yuexian Zou}, year = {2025}, url = {https://arxiv.org/abs/2409.09256}, note = {Source identifier: 2409.09256} }