@misc{indiciae12b6c4be5748, title = {Efficiently Identifying Low-Quality Language Subsets in Multilingual Datasets: A Case Study on a Large-Scale Multilingual Audio Dataset}, author = {Farhan Samir and Emily P. Ahn and Shreya Prakash and Márton Soskuthy and Vered Shwartz and Jian Zhu}, year = {2024}, url = {https://arxiv.org/abs/2410.04292}, note = {Source identifier: 2410.04292} }