@misc{indiciae2824f26a6d3c, title = {Generation-Step-Aware Framework for Cross-Modal Representation and Control in Multilingual Speech-Text Models}, author = {Toshiki Nakai and Varsha Suresh and Vera Demberg}, year = {2026}, url = {https://arxiv.org/abs/2601.17387}, note = {Source identifier: 2601.17387} }