@misc{indiciae53ef74996b60, title = {Multimodal Emotion Recognition using Audio-Video Transformer Fusion with Cross Attention}, author = {Joe Dhanith P R and Shravan Venkatraman and Vigya Sharma and Santhosh Malarvannan}, year = {2026}, url = {https://arxiv.org/abs/2407.18552}, note = {Source identifier: 2407.18552} }