@misc{indiciae2b6efe698278, title = {Hear What Matters! Text-conditioned Selective Video-to-Audio Generation}, author = {Junwon Lee and Juhan Nam and Jiyoung Lee}, year = {2026}, url = {https://arxiv.org/abs/2512.02650}, note = {Source identifier: 2512.02650} }