@misc{indiciae15e95665b109, title = {TMT: Tri-Modal Translation between Speech, Image, and Text by Processing Different Modalities as Different Languages}, author = {Minsu Kim and Jee-weon Jung and Hyeongseop Rha and Soumi Maiti and Siddhant Arora and Xuankai Chang and Shinji Watanabe and Yong Man Ro}, year = {2025}, url = {https://arxiv.org/abs/2402.16021}, note = {Source identifier: 2402.16021} }