@misc{indiciaea17d9bc64a48, title = {PixIT: Joint Training of Speaker Diarization and Speech Separation from Real-world Multi-speaker Recordings}, author = {Joonas Kalda and Clément Pagés and Ricard Marxer and Tanel Alumäe and Hervé Bredin}, year = {2024}, doi = {10.21437/odyssey.2024-17}, url = {https://arxiv.org/abs/2403.02288}, note = {Source identifier: 2403.02288} }