@misc{indiciae42b06d9b8a31, title = {Discrete Multimodal Transformers with a Pretrained Large Language Model for Mixed-Supervision Speech Processing}, author = {Viet Anh Trinh and Rosy Southwell and Yiwen Guan and Xinlu He and Zhiyong Wang and Jacob Whitehill}, year = {2024}, url = {https://arxiv.org/abs/2406.06582}, note = {Source identifier: 2406.06582} }