@misc{indiciaeed4f86b8053e, title = {When Metrics Disagree: Automatic Similarity vs. LLM-as-a-Judge for Clinical Dialogue Evaluation}, author = {Bian Sun and Zhenjian Wang and Orvill de la Torre and Zirui Wang}, year = {2026}, url = {https://arxiv.org/abs/2603.00314}, note = {Source identifier: 2603.00314} }