@misc{indiciaee392f2377e64, title = {AVID: A Benchmark for Omni-Modal Audio-Visual Inconsistency Understanding via Agent-Driven Construction}, author = {Zixuan Chen and Depeng Wang and Hao Lin and Li Luo and Ke Xu and Ya Guo and Huijia Zhu and Tanfeng Sun and Xinghao Jiang}, year = {2026}, url = {https://arxiv.org/abs/2604.13593}, note = {Source identifier: 2604.13593} }