@misc{indiciae09cee9575b6e, title = {Do Audio-Visual Large Language Models Really See and Hear?}, author = {Ramaneswaran Selvakumar and Kaousheik Jayakumar and S Sakshi and Sreyan Ghosh and Ruohan Gao and Dinesh Manocha}, year = {2026}, url = {https://arxiv.org/abs/2604.02605}, note = {Source identifier: 2604.02605} }