@misc{indiciaeb21e32a48ff9, title = {SceneBind: Binding What and Where Across Vision, Audio and Language}, author = {Mingfei Chen and Zijun Cui and Ruoke Zhang and Hyeonggon Ryu and Eli Shlizerman}, year = {2026}, url = {https://arxiv.org/abs/2607.15265}, note = {Source identifier: 2607.15265} }