@misc{indiciaed0ffd5223394, title = {Multimodal Speech Emotion Recognition using Cross Attention with Aligned Audio and Text}, author = {Yoonhyung Lee and Seunghyun Yoon and Kyomin Jung}, year = {2022}, doi = {10.21437/interspeech.2020-2312}, url = {https://arxiv.org/abs/2207.12895}, note = {Source identifier: 2207.12895} }