@misc{indiciae49684814729a, title = {Learning Audio-Visual Embeddings with Inferred Latent Interaction Graphs}, author = {Donghuo Zeng and Hao Niu and Yanan Wang and Masato Taya}, year = {2026}, url = {https://arxiv.org/abs/2601.11995}, note = {Source identifier: 2601.11995} }