@misc{indiciae75ff902b66f1, title = {It Hears, It Sees too: Multi-Modal LLM for Depression Detection By Integrating Visual Understanding into Audio Language Models}, author = {Xiangyu Zhao and Yaling Shen and Yiwen Jiang and Zimu Wang and Jiahe Liu and Maxmartwell H Cheng and Guilherme C Oliveira and Robert Desimone and Dominic Dwyer and Zongyuan Ge}, year = {2025}, url = {https://arxiv.org/abs/2511.19877}, note = {Source identifier: 2511.19877} }