@misc{indiciae4929e17202b4, title = {Robust Multi-Tier Infant-Centered Audio Understanding with Whisper via Structured Speaker Conditioning}, author = {Xulin Fan and Jialu Li and Mohammad Nur Hossain Khan and Kexin Hu and Bashima Islam and Mark Hasegawa-Johnson and Nancy L. McElwain}, year = {2026}, url = {https://arxiv.org/abs/2608.11587}, note = {Source identifier: 2608.11587} }