@misc{indiciae87b112c41e14, title = {Aligning Medical Conversational AI through Online Reinforcement Learning with Information-Theoretic Rewards}, author = {Tanvi Verma and Yang Zhou and Rick Siow Mong Goh and Yong Liu}, year = {2026}, url = {https://arxiv.org/abs/2601.17828}, note = {Source identifier: 2601.17828} }