@misc{indiciaed8ca12c30c9e, title = {Multifaceted Evaluation of Audio-Visual Capability for MLLMs: Effectiveness, Efficiency, Generalizability and Robustness}, author = {Yusheng Zhao and Junyu Luo and Xiao Luo and Weizhi Zhang and Zhiping Xiao and Wei Ju and Philip S. Yu and Ming Zhang}, year = {2025}, url = {https://arxiv.org/abs/2504.16936}, note = {Source identifier: 2504.16936} }