@misc{indiciae1085bd96eb8d, title = {Efficient Audio-Visual Generation via Synchrony-Aware Cross-Modal Sparse Attention}, author = {Shengchuan Gao and Teng Hu and Bohao Feng and Luchen Li and Wenqiang Wang and Hongqian Deng and Ran Yi}, year = {2026}, url = {https://arxiv.org/abs/2608.15522}, note = {Source identifier: 2608.15522} }