@misc{indiciaedaed93916529, title = {Voice Activity Projection Model with Multimodal Encoders}, author = {Takeshi Saga and Catherine Pelachaud}, year = {2025}, url = {https://arxiv.org/abs/2506.03980}, note = {Source identifier: 2506.03980} }