@misc{indiciaeb58eec47811f, title = {Style-transfer based Speech and Audio-visual Scene Understanding for Robot Action Sequence Acquisition from Videos}, author = {Chiori Hori and Puyuan Peng and David Harwath and Xinyu Liu and Kei Ota and Siddarth Jain and Radu Corcodel and Devesh Jha and Diego Romeres and Jonathan Le Roux}, year = {2023}, url = {https://arxiv.org/abs/2306.15644}, note = {Source identifier: 2306.15644} }