@misc{indiciaeec6e6ae8ebb7, title = {CAT: Enhancing Multimodal Large Language Model to Answer Questions in Dynamic Audio-Visual Scenarios}, author = {Qilang Ye and Zitong Yu and Rui Shao and Xinyu Xie and Philip Torr and Xiaochun Cao}, year = {2024}, url = {https://arxiv.org/abs/2403.04640}, note = {Source identifier: 2403.04640} }